singam96 commited on
Commit
2d1eb0d
·
verified ·
1 Parent(s): e8d4d21

ShadeNet-3.2-5M (model): weights, ONNX, app + card

Browse files
Files changed (41) hide show
  1. .gitattributes +25 -0
  2. README.md +203 -0
  3. app.py +59 -0
  4. assets/103195344_5d2dc613a3_result.png +3 -0
  5. assets/154871781_ae77696b77_result.png +3 -0
  6. assets/1561658940_a947f2446a_result.png +3 -0
  7. assets/157139628_5dc483e2e4_result.png +3 -0
  8. assets/159712188_d530dd478c_result.png +3 -0
  9. assets/160541986_d5be2ab4c1_result.png +3 -0
  10. assets/160566014_59528ff897_result.png +3 -0
  11. assets/160585932_fa6339f248_result.png +3 -0
  12. assets/160792599_6a7ec52516_result.png +3 -0
  13. assets/161669933_3e7d8c7e2c_result.png +3 -0
  14. assets/2260560631_09093be4c6_result.png +3 -0
  15. assets/2312984882_bec7849e09_result.png +3 -0
  16. assets/241345721_3f3724a7fc_result.png +3 -0
  17. assets/2453318633_550228acd4_result.png +3 -0
  18. assets/252578659_9e404b6430_result.png +3 -0
  19. assets/307994435_592f933a6d_result.png +3 -0
  20. assets/3185645793_49de805194_result.png +3 -0
  21. assets/326585030_e1dcca2562_result.png +3 -0
  22. assets/3440104178_6871a24e13_result.png +3 -0
  23. assets/487071033_27e460a1b9_result.png +3 -0
  24. assets/atom_dictionary.png +3 -0
  25. assets/compare_154871781_ae77696b77.png +3 -0
  26. assets/compare_159712188_d530dd478c.png +3 -0
  27. assets/compare_160585932_fa6339f248.png +3 -0
  28. assets/examples/154871781_ae77696b77.jpg +0 -0
  29. assets/examples/159712188_d530dd478c.jpg +0 -0
  30. assets/examples/160585932_fa6339f248.jpg +0 -0
  31. assets/hero.png +3 -0
  32. assets/training_curves.png +0 -0
  33. checkpoints/shadenet32.ckpt +3 -0
  34. config.json +18 -0
  35. inference.py +68 -0
  36. inference_utils.py +102 -0
  37. model.py +338 -0
  38. onnx/model.onnx +3 -0
  39. onnx/model_fp16.onnx +3 -0
  40. requirements-space.txt +5 -0
  41. requirements.txt +6 -0
.gitattributes CHANGED
@@ -33,3 +33,28 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ assets/103195344_5d2dc613a3_result.png filter=lfs diff=lfs merge=lfs -text
37
+ assets/154871781_ae77696b77_result.png filter=lfs diff=lfs merge=lfs -text
38
+ assets/1561658940_a947f2446a_result.png filter=lfs diff=lfs merge=lfs -text
39
+ assets/157139628_5dc483e2e4_result.png filter=lfs diff=lfs merge=lfs -text
40
+ assets/159712188_d530dd478c_result.png filter=lfs diff=lfs merge=lfs -text
41
+ assets/160541986_d5be2ab4c1_result.png filter=lfs diff=lfs merge=lfs -text
42
+ assets/160566014_59528ff897_result.png filter=lfs diff=lfs merge=lfs -text
43
+ assets/160585932_fa6339f248_result.png filter=lfs diff=lfs merge=lfs -text
44
+ assets/160792599_6a7ec52516_result.png filter=lfs diff=lfs merge=lfs -text
45
+ assets/161669933_3e7d8c7e2c_result.png filter=lfs diff=lfs merge=lfs -text
46
+ assets/2260560631_09093be4c6_result.png filter=lfs diff=lfs merge=lfs -text
47
+ assets/2312984882_bec7849e09_result.png filter=lfs diff=lfs merge=lfs -text
48
+ assets/241345721_3f3724a7fc_result.png filter=lfs diff=lfs merge=lfs -text
49
+ assets/2453318633_550228acd4_result.png filter=lfs diff=lfs merge=lfs -text
50
+ assets/252578659_9e404b6430_result.png filter=lfs diff=lfs merge=lfs -text
51
+ assets/307994435_592f933a6d_result.png filter=lfs diff=lfs merge=lfs -text
52
+ assets/3185645793_49de805194_result.png filter=lfs diff=lfs merge=lfs -text
53
+ assets/326585030_e1dcca2562_result.png filter=lfs diff=lfs merge=lfs -text
54
+ assets/3440104178_6871a24e13_result.png filter=lfs diff=lfs merge=lfs -text
55
+ assets/487071033_27e460a1b9_result.png filter=lfs diff=lfs merge=lfs -text
56
+ assets/atom_dictionary.png filter=lfs diff=lfs merge=lfs -text
57
+ assets/compare_154871781_ae77696b77.png filter=lfs diff=lfs merge=lfs -text
58
+ assets/compare_159712188_d530dd478c.png filter=lfs diff=lfs merge=lfs -text
59
+ assets/compare_160585932_fa6339f248.png filter=lfs diff=lfs merge=lfs -text
60
+ assets/hero.png filter=lfs diff=lfs merge=lfs -text
README.md CHANGED
@@ -1,3 +1,206 @@
1
  ---
2
  license: apache-2.0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3
  ---
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
  license: apache-2.0
3
+ pipeline_tag: image-to-image
4
+ library_name: pytorch
5
+ tags:
6
+ - onnx
7
+ - inverse-rendering
8
+ - image-decomposition
9
+ - albedo
10
+ - normal-map
11
+ - depth-estimation
12
+ - pbr
13
+ - material-estimation
14
+ - flickr8k
15
+ - shadenet
16
+ datasets:
17
+ - singam96/flickr8k_marigold_v2
18
+ metrics:
19
+ - mse
20
+ model-index:
21
+ - name: ShadeNet-3.2-5M
22
+ results:
23
+ - task:
24
+ type: image-to-image
25
+ dataset:
26
+ name: flickr8k_marigold_v2 (val split, 807 images)
27
+ type: singam96/flickr8k_marigold_v2
28
+ metrics:
29
+ - name: val/loss (weighted MSE + recon)
30
+ type: mse
31
+ value: 0.1671
32
  ---
33
+
34
+ # ShadeNet-3.2 5M
35
+
36
+ A lightweight inverse-rendering model: one photo in, **albedo + relative depth + surface normals + shading** out (8 channels, 5.0M params). Successor of [ShadeNet-2](https://huggingface.co/singam96/ShadeNet-2-20M) — **4× smaller, better depth and normals**.
37
+
38
+ [![hero](assets/hero.png)](assets/hero.png)
39
+
40
+ ## TL;DR
41
+
42
+ | | |
43
+ |---|---|
44
+ | Params | **5.0M** (3.2M trainable + 1.8M frozen MobileNetV2 trunk) |
45
+ | Input | RGB `[1, 3, H, W]` in `[-1, 1]` (384px trained; any multiple of 16) |
46
+ | Output | 8ch `[1, 8, H, W]` in `[-1, 1]`: albedo, relative depth, normals, shading |
47
+ | Val L1 (807 imgs) | albedo 0.695 · depth 0.217 · normal 0.581 |
48
+ | Formats | fp32 ONNX (20MB), fp16 ONNX (10MB), torch checkpoint (44MB) |
49
+
50
+ ## Examples
51
+
52
+ [![154871781](assets/154871781_ae77696b77_result.png)](assets/154871781_ae77696b77_result.png)
53
+ [![157139628](assets/157139628_5dc483e2e4_result.png)](assets/157139628_5dc483e2e4_result.png)
54
+ [![159712188](assets/159712188_d530dd478c_result.png)](assets/159712188_d530dd478c_result.png)
55
+ [![160541986](assets/160541986_d5be2ab4c1_result.png)](assets/160541986_d5be2ab4c1_result.png)
56
+ [![160566014](assets/160566014_59528ff897_result.png)](assets/160566014_59528ff897_result.png)
57
+ [![160585932](assets/160585932_fa6339f248_result.png)](assets/160585932_fa6339f248_result.png)
58
+ [![160792599](assets/160792599_6a7ec52516_result.png)](assets/160792599_6a7ec52516_result.png)
59
+ [![161669933](assets/161669933_3e7d8c7e2c_result.png)](assets/161669933_3e7d8c7e2c_result.png)
60
+
61
+ *Each grid: input | albedo | shading / depth | normal | recon (albedo×shading). Click any image for full size.*
62
+
63
+ ## Results
64
+
65
+ Full 807-image val split, per-map L1 (the comparable metric across versions —
66
+ the headline `val/loss` formula changed between v2 and v3):
67
+
68
+ | Map (val L1) | ShadeNet-2 (20M) | **ShadeNet-3.2 (5M)** | change |
69
+ |---|---|---|---|
70
+ | Albedo | 0.708 | **0.695** | −1.7% |
71
+ | Depth (SSI-aligned) | 0.247 | **0.217** | **−12%** |
72
+ | Normal | 0.696 | **0.581** | **−16%** |
73
+
74
+ [![curves](assets/training_curves.png)](assets/training_curves.png)
75
+
76
+ Checkpoint variants (full val, pruned top-32 dictionary):
77
+
78
+ | Weights | val/loss | albedo L1 | depth L1 | normal L1 |
79
+ |---|---|---|---|---|
80
+ | best.ckpt (raw) | **0.1669** | 0.7013 | 0.2241 | **0.5709** |
81
+ | best.ckpt (EMA) | 0.1671 | **0.6952** | **0.2174** | 0.5807 |
82
+
83
+ Shipped ONNX uses the **EMA** weights. All page outputs and the Space run a
84
+ **3-pass multi-scale median** (scales 0.875/1.0/1.125) — a mild denoise; the
85
+ model's raw single-pass albedo is sharper than its pseudo-labels, so this
86
+ trades a little detail for lower variance.
87
+
88
+ ## ShadeNet-2 vs ShadeNet-3.2
89
+
90
+ Same inputs (ShadeNet-2 top rows, ShadeNet-3.2 bottom rows), maps only. Both
91
+ use their shipped weights; ShadeNet-3.2 runs the 3-pass multi-scale median.
92
+
93
+ [![compare 1](assets/compare_154871781_ae77696b77.png)](assets/compare_154871781_ae77696b77.png)
94
+ [![compare 2](assets/compare_159712188_d530dd478c.png)](assets/compare_159712188_d530dd478c.png)
95
+ [![compare 3](assets/compare_160585932_fa6339f248.png)](assets/compare_160585932_fa6339f248.png)
96
+
97
+ ## Architecture
98
+
99
+ **ParallelUNet generator (4.98M params)** + spectral-norm GroupNorm PatchGAN discriminator (2.77M, training only):
100
+
101
+ - Dual parallel encoders — vanilla UNet path plus a **frozen MobileNetV2** feature trunk, fused at every decoder level
102
+ - **Depthwise-separable factorized convs** (1×3 + 3×1) throughout; full H/32 bottleneck; reflect padding
103
+ - **Patch-dictionary output tail**: 16×16 tiles softmax-addressed over **32 learned per-channel atoms** (pruned from 1024 — the top-32 hold 99.4% of addressing mass), blended back into the signal before tanh
104
+ - Single-pass RGB → 8ch output; EMA weight shadow (shipped weights are EMA)
105
+
106
+ ## Patch dictionary
107
+
108
+ The tail softmax-addresses 32 learned 16×16 atoms per tile (kept from 1024
109
+ after measuring per-atom selection: only ~34 atoms are ever used, ~31 cover
110
+ 99% of the mass). Shown below per output channel — each panel is the 8×4 atom
111
+ grid, shared grayscale scale.
112
+
113
+ [![atom dictionary](assets/atom_dictionary.png)](assets/atom_dictionary.png)
114
+
115
+ ## Output maps
116
+
117
+ | Map | Channels | Range | Description |
118
+ |---|---|---|---|
119
+ | Albedo | 3 `[0:3]` | [−1, 1] | Reflectance / diffuse color, lighting factored out |
120
+ | Depth | 1 `[3:4]` | [−1, 1] | **Relative** depth (0=near), affine-ambiguous |
121
+ | Normal | 3 `[4:7]` | [−1, 1] | Surface normals, unit-length regularised |
122
+ | Shading | 1 `[7:8]` | [−1, 1] | Grayscale irradiance; `input ≈ albedo × shading` |
123
+ | Recon | — | — | `albedo × shading` re-rendering (diagnostic, not a head) |
124
+
125
+ ## Files
126
+
127
+ ```
128
+ ├── app.py # Gradio Space app (fp16 ONNX)
129
+ ├── inference.py # Standalone torch CLI
130
+ ├── inference_utils.py # Grid visualisation (numpy/PIL)
131
+ ├── model.py # Standalone generator architecture
132
+ ├── requirements.txt
133
+ ├── checkpoints/shadenet32.ckpt # Torch weights, EMA (44MB)
134
+ └── onnx/
135
+ ├── model.onnx # fp32, EMA (20MB) — GPU via CUDA EP
136
+ └── model_fp16.onnx # fp16, EMA (10MB) — CPU
137
+ ```
138
+
139
+ ## Usage
140
+
141
+ ### Gradio Space
142
+
143
+ Try it in your browser — no installation: **[singam96/ShadeNet-3.2-5M Space](https://huggingface.co/spaces/singam96/ShadeNet-3.2-5M)**.
144
+
145
+ This repo ships `app.py`, the Space entrypoint. To recreate it: New Space → Gradio SDK → point at this repo.
146
+
147
+ ### Torch CLI
148
+
149
+ ```bash
150
+ pip install torch torchvision pillow numpy
151
+ python inference.py photo.jpg --output_dir ./output
152
+ # --checkpoint ./checkpoints/shadenet32.ckpt --image-size 512 --no-ema to disable EMA
153
+ ```
154
+
155
+ ### ONNX (CPU)
156
+
157
+ ```bash
158
+ pip install onnxruntime pillow numpy
159
+ python - <<'EOF'
160
+ import onnxruntime as ort, numpy as np
161
+ from PIL import Image
162
+ from inference_utils import build_grid, pil_to_np, resize_pad
163
+ sess = ort.InferenceSession("onnx/model_fp16.onnx", providers=["CPUExecutionProvider"])
164
+ img = resize_pad(Image.open("photo.jpg"), 512)
165
+ out = sess.run(None, {"input_rgb": pil_to_np(img).astype(np.float32)})[0]
166
+ build_grid(img, out).save("result.png")
167
+ EOF
168
+ ```
169
+
170
+ Input: `[1, 3, H, W]` in `[-1, 1]` (any H, W; multiples of 16 recommended).
171
+ Output: `[1, 8, H, W]` in `[-1, 1]`.
172
+
173
+ ## Training
174
+
175
+ Trained from scratch on [`singam96/flickr8k_marigold_v2`](https://huggingface.co/datasets/singam96/flickr8k_marigold_v2) (8077 Flickr8k photos with Marigold-V2 pseudo-labels), 384px, fp32, single GTX 1650, early-stopped on `val/loss` (patience 5) at epoch 12. The 1024-atom dictionary was pruned to its top-32 by addressing mass (no retraining; output error vs full ~5e-4 mean) for the release.
176
+
177
+ Losses: scale-invariant MSE on albedo (per-channel std alignment) + **scale-shift-invariant** MSE on depth after least-squares alignment (decoded Marigold depth is relative) + **Sobel gradient-matching** on depth (edge crispness) + MSE on normals + self-supervised **reconstruction coupling** (`albedo×shading ≈ input`, the shading head's only supervision) + LSGAN + normal unit-length penalty. Weight decay 1e-4 (patch dictionary exempt).
178
+
179
+ ## Limitations
180
+
181
+ - Depth is **relative**, not metric — don't read meters off it.
182
+ - Shading assumes **white light**; strongly colored illumination (sunsets, neon) leaks into albedo.
183
+ - Normals are noisy in foliage/sky — those pseudo-labels were noisy too.
184
+ - Occasional localized artifacts in albedo/shading (learned prior pockets).
185
+ - No shadows/global illumination — relighting-style use is approximate.
186
+ - Uncertainty is not provided: confidently-wrong pseudo-labels are fitted confidently.
187
+
188
+ ## Attribution
189
+
190
+ Supervision labels come from **Marigold V2** (Ke et al.) applied to **Flickr8k** (Hodosh et al.):
191
+
192
+ - Marigold: *Repurposing Diffusion-Based Image Generators for Monocular Depth Estimation* — Ke, Obukhov, Metzger, Daudt, Schindler, Schindler (CVPR 2024)
193
+ - Flickr8k: *Framing Image Description as a Ranking Task* — Hodosh, Young, Hockenmaier (2013)
194
+
195
+ This model (weights + code) is Apache-2.0; upstream dataset/model terms still apply to their artifacts.
196
+
197
+ ## Citation
198
+
199
+ ```bibtex
200
+ @software{shadenet32,
201
+ author = {Sachin},
202
+ title = {ShadeNet-3.2: single-image inverse rendering (5M)},
203
+ year = {2026},
204
+ url = {https://huggingface.co/singam96/ShadeNet-3.2-5M}
205
+ }
206
+ ```
app.py ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """ShadeNet-3.2 Gradio Space (CPU-friendly: fp16 ONNX backend)."""
2
+ import os
3
+
4
+ import gradio as gr
5
+ import numpy as np
6
+ import onnxruntime as ort
7
+ import spaces
8
+ from PIL import Image
9
+
10
+ from inference_utils import (build_grid, extract_maps, pil_to_np, resize_pad,
11
+ run_ensemble_onnx)
12
+
13
+ MODEL_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)),
14
+ "onnx", "model_fp16.onnx")
15
+ IMAGE_SIZE = 512
16
+
17
+ _sess = ort.InferenceSession(MODEL_PATH, providers=["CPUExecutionProvider"])
18
+
19
+
20
+ @spaces.GPU(duration=60)
21
+ def decompose(image: Image.Image):
22
+ """Returns (grid, input, albedo, shading, depth, normal, recon)."""
23
+ img_rgb = resize_pad(image.convert("RGB"), IMAGE_SIZE)
24
+ out = run_ensemble_onnx(_sess, img_rgb, IMAGE_SIZE)
25
+ maps = extract_maps(out, img_rgb)
26
+ return (build_grid(img_rgb, out), maps["input"], maps["albedo"],
27
+ maps["shading"], maps["depth"], maps["normal"], maps["recon"])
28
+
29
+
30
+ EXAMPLES = []
31
+ _exdir = os.path.join(os.path.dirname(os.path.abspath(__file__)),
32
+ "assets", "examples")
33
+ if os.path.isdir(_exdir):
34
+ EXAMPLES = sorted(os.path.join(_exdir, f) for f in os.listdir(_exdir)
35
+ if f.lower().endswith((".png", ".jpg", ".jpeg")))
36
+
37
+ demo = gr.Interface(
38
+ fn=decompose,
39
+ inputs=gr.Image(type="pil", label="Input photo"),
40
+ outputs=[
41
+ gr.Image(type="pil", label="Overview grid"),
42
+ gr.Image(type="pil", label="Input (resized)"),
43
+ gr.Image(type="pil", label="Albedo"),
44
+ gr.Image(type="pil", label="Shading"),
45
+ gr.Image(type="pil", label="Depth (relative: dark=near)"),
46
+ gr.Image(type="pil", label="Normal"),
47
+ gr.Image(type="pil", label="Recon (albedo x shading)"),
48
+ ],
49
+ title="ShadeNet-3.2: single-image inverse rendering (7M)",
50
+ description=("Decomposes a photo into albedo, relative depth, surface "
51
+ "normals and shading (grayscale irradiance). Successor of "
52
+ "ShadeNet-2 (20M), 3x smaller with better depth/normal "
53
+ "accuracy. Runs the fp16 ONNX model."),
54
+ examples=EXAMPLES,
55
+ cache_examples=False,
56
+ )
57
+
58
+ if __name__ == "__main__":
59
+ demo.launch()
assets/103195344_5d2dc613a3_result.png ADDED

Git LFS Details

  • SHA256: c38b619fa1c835b07bb6344539d264bb0ba15164a65b4a7e08dd9410948b4d23
  • Pointer size: 131 Bytes
  • Size of remote file: 523 kB
assets/154871781_ae77696b77_result.png ADDED

Git LFS Details

  • SHA256: b12ef6fd1cc407bc033fddb3c4c28f98cc1a3ee8edcf0e367b7342238f0a5b95
  • Pointer size: 131 Bytes
  • Size of remote file: 618 kB
assets/1561658940_a947f2446a_result.png ADDED

Git LFS Details

  • SHA256: bbf9f5297f14721bf9e94673d9a7c88c35e0f88e67b0666a6022fa01817cc7eb
  • Pointer size: 131 Bytes
  • Size of remote file: 682 kB
assets/157139628_5dc483e2e4_result.png ADDED

Git LFS Details

  • SHA256: 2c9f8574508263ce4baf15e71d5a7e5d1b11df5a44a4f42e9f0e19da890a4d7d
  • Pointer size: 131 Bytes
  • Size of remote file: 473 kB
assets/159712188_d530dd478c_result.png ADDED

Git LFS Details

  • SHA256: 74cd8d9f3b15fdcf6100fd276bfa1cdf903978d465432d44df0a5d80799fbec7
  • Pointer size: 131 Bytes
  • Size of remote file: 527 kB
assets/160541986_d5be2ab4c1_result.png ADDED

Git LFS Details

  • SHA256: 8a7326efd27c021019d75d5c8aaebda88efa10ad8f595d045eb90847e1775818
  • Pointer size: 131 Bytes
  • Size of remote file: 535 kB
assets/160566014_59528ff897_result.png ADDED

Git LFS Details

  • SHA256: 3c27df341a578f06e50a722610814b0c179bdd87edcd764c95550117440f8d0a
  • Pointer size: 131 Bytes
  • Size of remote file: 566 kB
assets/160585932_fa6339f248_result.png ADDED

Git LFS Details

  • SHA256: 14b0eaac4b217b135f455012de606356a59aad81e9b5e4abe4d6ea6abca1052f
  • Pointer size: 131 Bytes
  • Size of remote file: 666 kB
assets/160792599_6a7ec52516_result.png ADDED

Git LFS Details

  • SHA256: 5afe623f7b54bbff14e27d8ab96ba5492203c52e1b0192f7d3238b0b462bdb0d
  • Pointer size: 131 Bytes
  • Size of remote file: 496 kB
assets/161669933_3e7d8c7e2c_result.png ADDED

Git LFS Details

  • SHA256: 0303f28bfc8366b94253ae7da1884b20a081bfe76a73ef890163d1a5befa0aa2
  • Pointer size: 131 Bytes
  • Size of remote file: 529 kB
assets/2260560631_09093be4c6_result.png ADDED

Git LFS Details

  • SHA256: 2c2789566ba6e49dffce6eca75c5840d32474f370523d22c2c8ec6568f9ccd39
  • Pointer size: 131 Bytes
  • Size of remote file: 629 kB
assets/2312984882_bec7849e09_result.png ADDED

Git LFS Details

  • SHA256: 4969210d049e9a5047808b8964fe1c7b33d2941b70b25e38c433010f08b3adcb
  • Pointer size: 131 Bytes
  • Size of remote file: 567 kB
assets/241345721_3f3724a7fc_result.png ADDED

Git LFS Details

  • SHA256: 7c6877746e7981d2478ddc13f58833cf1fe2107cd1fbae69096ff47899044aca
  • Pointer size: 131 Bytes
  • Size of remote file: 554 kB
assets/2453318633_550228acd4_result.png ADDED

Git LFS Details

  • SHA256: 72a78491f60ef9b1f419c2adb61bd9a2fc69c4f98c377f57c7dfa662158729d3
  • Pointer size: 131 Bytes
  • Size of remote file: 526 kB
assets/252578659_9e404b6430_result.png ADDED

Git LFS Details

  • SHA256: 4e1fd8c37bdddf509f0c4f289a990a32f7086d4b9a1297ea0e380777381049dc
  • Pointer size: 131 Bytes
  • Size of remote file: 565 kB
assets/307994435_592f933a6d_result.png ADDED

Git LFS Details

  • SHA256: c2b8e3cda9e10e52a56419a3a633465f4dfe5797673d170e4a587c0cfe00bf5c
  • Pointer size: 131 Bytes
  • Size of remote file: 553 kB
assets/3185645793_49de805194_result.png ADDED

Git LFS Details

  • SHA256: aa75ba3cfd30d30486dbdaca36130532f2e2f863a2c57a5229cadf774db557dd
  • Pointer size: 131 Bytes
  • Size of remote file: 616 kB
assets/326585030_e1dcca2562_result.png ADDED

Git LFS Details

  • SHA256: 43c6770ec06f641ad50336263247b283e58bd0d5bc8117c85f83bc9a119008a4
  • Pointer size: 131 Bytes
  • Size of remote file: 518 kB
assets/3440104178_6871a24e13_result.png ADDED

Git LFS Details

  • SHA256: 28bbc6284dc68405ec4244940b417baeb5c3f1cabd6088db1a39e1ce14ea67b0
  • Pointer size: 131 Bytes
  • Size of remote file: 516 kB
assets/487071033_27e460a1b9_result.png ADDED

Git LFS Details

  • SHA256: 56423c3c48234d18922fc6b14427b290b7aec214dea34dc0c544df28050b6923
  • Pointer size: 131 Bytes
  • Size of remote file: 583 kB
assets/atom_dictionary.png ADDED

Git LFS Details

  • SHA256: 1a37d71c66b01385c2b817834b2833180eb68a643f2c5aa9914bd13a435530a8
  • Pointer size: 131 Bytes
  • Size of remote file: 134 kB
assets/compare_154871781_ae77696b77.png ADDED

Git LFS Details

  • SHA256: 5a005b3f3ba5aad666067a7079a504646fd91881aadbe704acabe928944add91
  • Pointer size: 131 Bytes
  • Size of remote file: 351 kB
assets/compare_159712188_d530dd478c.png ADDED

Git LFS Details

  • SHA256: 8763a52c3adeeda8a2ee1f56b02bf761291617b841829b6622b83284b73e17d7
  • Pointer size: 131 Bytes
  • Size of remote file: 308 kB
assets/compare_160585932_fa6339f248.png ADDED

Git LFS Details

  • SHA256: ffc13a404bc01479548bc8ffe2ad59c4f3dacf7c98281073099a0ab4dc40b5c6
  • Pointer size: 131 Bytes
  • Size of remote file: 368 kB
assets/examples/154871781_ae77696b77.jpg ADDED
assets/examples/159712188_d530dd478c.jpg ADDED
assets/examples/160585932_fa6339f248.jpg ADDED
assets/hero.png ADDED

Git LFS Details

  • SHA256: 10711621703e84a2841db8cd5902af5f8d014a66cfc616ee348023e2bcd1656d
  • Pointer size: 132 Bytes
  • Size of remote file: 4.6 MB
assets/training_curves.png ADDED
checkpoints/shadenet32.ckpt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3b660a6d955f2df63304312d0fe491bb6a0bd9e71fcdff69a7398a9b19fe3fd0
3
+ size 44034045
config.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model_type": "shadenet",
3
+ "model_name": "ShadeNet-3.2-5M",
4
+ "architecture": "ParallelUNet-v3",
5
+ "params_total": 4977208,
6
+ "params_trainable": 3165496,
7
+ "image_size": 384,
8
+ "in_ch": 3,
9
+ "out_ch": 8,
10
+ "maps": {
11
+ "albedo": [0, 3],
12
+ "depth": [3, 4],
13
+ "normal": [4, 7],
14
+ "shading": [7, 8]
15
+ },
16
+ "framework": "pytorch",
17
+ "onnx": ["onnx/model.onnx", "onnx/model_fp16.onnx"]
18
+ }
inference.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """ShadeNet-3.2 torch inference (standalone CLI).
2
+
3
+ Usage:
4
+ pip install -r requirements.txt
5
+ python inference.py input.jpg --output_dir ./output
6
+ python inference.py ./photos --checkpoint ./checkpoints/shadenet32.ckpt
7
+ """
8
+ import argparse
9
+ import os
10
+
11
+ import torch
12
+ from PIL import Image
13
+
14
+ from model import load_shadenet32
15
+ from inference_utils import build_grid, pil_to_np, resize_pad
16
+
17
+ IMAGE_SIZE = 512
18
+ ENS_SCALES = (0.875, 1.0, 1.125)
19
+
20
+
21
+ @torch.no_grad()
22
+ def run(model, img_rgb: Image.Image, device: str,
23
+ image_size: int = IMAGE_SIZE, scales=ENS_SCALES) -> "torch.Tensor":
24
+ """3-pass multi-scale median (matches the model card / Space)."""
25
+ outs = []
26
+ for s in scales:
27
+ side = max(16, int(round(image_size * s)) // 16 * 16) # dict needs /16
28
+ scaled = resize_pad(img_rgb, side)
29
+ x = torch.from_numpy(pil_to_np(scaled)).to(device)
30
+ o = model(x)
31
+ if side != image_size:
32
+ o = torch.nn.functional.interpolate(
33
+ o, size=(image_size, image_size), mode="bilinear",
34
+ align_corners=False)
35
+ outs.append(o.cpu())
36
+ return torch.stack(outs, dim=0).median(dim=0).values
37
+
38
+
39
+ def main():
40
+ ap = argparse.ArgumentParser(description="ShadeNet-3.2 inference")
41
+ ap.add_argument("input", help="Image file or folder")
42
+ ap.add_argument("--checkpoint", default="./checkpoints/shadenet32.ckpt")
43
+ ap.add_argument("--output_dir", default="./output")
44
+ ap.add_argument("--image-size", type=int, default=IMAGE_SIZE)
45
+ ap.add_argument("--no-ema", action="store_true",
46
+ help="Use raw weights instead of EMA")
47
+ args = ap.parse_args()
48
+
49
+ device = "cuda" if torch.cuda.is_available() else "cpu"
50
+ model = load_shadenet32(args.checkpoint, device, use_ema=not args.no_ema)
51
+
52
+ if os.path.isdir(args.input):
53
+ files = sorted(os.path.join(args.input, f) for f in os.listdir(args.input)
54
+ if f.lower().endswith((".png", ".jpg", ".jpeg", ".webp")))
55
+ else:
56
+ files = [args.input]
57
+ os.makedirs(args.output_dir, exist_ok=True)
58
+ for fp in files:
59
+ img_rgb = resize_pad(Image.open(fp), args.image_size)
60
+ out = run(model, img_rgb, device).numpy()
61
+ base = os.path.splitext(os.path.basename(fp))[0]
62
+ build_grid(img_rgb, out).save(
63
+ os.path.join(args.output_dir, f"{base}_result.png"))
64
+ print(f"Saved: {base}_result.png")
65
+
66
+
67
+ if __name__ == "__main__":
68
+ main()
inference_utils.py ADDED
@@ -0,0 +1,102 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Standalone inference utilities (numpy/PIL only — no torch needed)."""
2
+ import numpy as np
3
+ from PIL import Image, ImageDraw
4
+
5
+ ALBEDO = slice(0, 3)
6
+ DEPTH = slice(3, 4)
7
+ NORMAL = slice(4, 7)
8
+ SHADING = slice(7, 8)
9
+
10
+
11
+ def resize_pad(img_rgb: Image.Image, image_size: int) -> Image.Image:
12
+ """Scale + center-crop to a square (RGB in, RGB out)."""
13
+ img_rgb = img_rgb.convert("RGB")
14
+ w, h = img_rgb.size
15
+ size = int(image_size)
16
+ scale = max(size / float(w), size / float(h))
17
+ img_r = img_rgb.resize((max(1, int(round(w * scale))),
18
+ max(1, int(round(h * scale)))), Image.BICUBIC)
19
+ left = (img_r.size[0] - size) // 2
20
+ top = (img_r.size[1] - size) // 2
21
+ return img_r.crop((left, top, left + size, top + size))
22
+
23
+
24
+ def pil_to_np(img_rgb: Image.Image) -> np.ndarray:
25
+ """PIL RGB -> [1, 3, H, W] float32 in [-1, 1]."""
26
+ arr = np.array(img_rgb.convert("RGB"), dtype=np.float32).transpose(2, 0, 1)
27
+ return arr[np.newaxis] / 255.0 * 2.0 - 1.0
28
+
29
+
30
+ def _to_01(output: np.ndarray) -> np.ndarray:
31
+ return np.clip((output + 1.0) / 2.0, 0, 1)
32
+
33
+
34
+ def _tile(arr01: np.ndarray, ts: int) -> Image.Image:
35
+ return Image.fromarray(
36
+ (arr01 * 255.0 + 0.5).astype("uint8")).resize((ts, ts), Image.BICUBIC)
37
+
38
+
39
+ def _resize_map(out: np.ndarray, size: int) -> np.ndarray:
40
+ """Bilinear-resize a [1, C, H, W] float map to [1, C, size, size]."""
41
+ c = out.shape[1]
42
+ res = np.empty((1, c, size, size), dtype=np.float32)
43
+ for i in range(c):
44
+ im = Image.fromarray(out[0, i].astype(np.float32), mode="F")
45
+ res[0, i] = np.asarray(im.resize((size, size), Image.BILINEAR))
46
+ return res
47
+
48
+
49
+ ENS_SCALES = (0.875, 1.0, 1.125)
50
+
51
+
52
+ def run_ensemble_onnx(sess, img_rgb: Image.Image, image_size: int = 512,
53
+ scales=ENS_SCALES) -> np.ndarray:
54
+ """3-pass multi-scale median (torch-free). Returns [1, 8, S, S]."""
55
+ outs = []
56
+ for s in scales:
57
+ side = max(16, int(round(image_size * s)) // 16 * 16) # dict needs /16
58
+ scaled = resize_pad(img_rgb, side)
59
+ x = pil_to_np(scaled).astype(np.float32)
60
+ o = sess.run(None, {"input_rgb": x})[0]
61
+ outs.append(o if side == image_size else _resize_map(o, image_size))
62
+ return np.median(np.stack(outs, axis=0), axis=0).astype(np.float32)
63
+
64
+
65
+ def extract_maps(output: np.ndarray, img_rgb: Image.Image,
66
+ ts: int = 256) -> dict:
67
+ """Split a [1, 8, H, W] output ([-1, 1]) into labelled PIL tiles."""
68
+ assert output.shape[1] == 8, f"want 8 channels, got {output.shape[1]}"
69
+ o = _to_01(output)
70
+ alb = o[:, ALBEDO][0].transpose(1, 2, 0)
71
+ sh = o[:, SHADING][0, 0]
72
+ d = o[:, DEPTH][0, 0]
73
+ span = d.max() - d.min()
74
+ dg = (d - d.min()) / (span if span > 1e-6 else 1.0)
75
+ return {
76
+ "input": img_rgb.resize((ts, ts), Image.BICUBIC),
77
+ "albedo": _tile(alb, ts),
78
+ "shading": _tile(np.repeat(sh[..., None], 3, axis=2), ts),
79
+ "depth": _tile(dg, ts).convert("RGB"),
80
+ "normal": _tile(o[:, NORMAL][0].transpose(1, 2, 0), ts),
81
+ "recon": _tile(np.clip(alb * sh[..., None], 0, 1), ts),
82
+ }
83
+
84
+
85
+ def build_grid(img_rgb: Image.Image, output: np.ndarray,
86
+ ts: int = 256) -> Image.Image:
87
+ """3x2 labelled grid: input|albedo|shading / depth|normal|recon."""
88
+ maps = extract_maps(output, img_rgb, ts)
89
+ grid = Image.new("RGB", (ts * 3, ts * 2))
90
+ grid.paste(maps["input"], (0, 0))
91
+ grid.paste(maps["albedo"], (ts, 0))
92
+ grid.paste(maps["shading"], (ts * 2, 0))
93
+ grid.paste(maps["depth"], (0, ts))
94
+ grid.paste(maps["normal"], (ts, ts))
95
+ grid.paste(maps["recon"], (ts * 2, ts))
96
+ draw = ImageDraw.Draw(grid)
97
+ for (x, y, name) in [(4, 4, "input"), (ts + 4, 4, "albedo"),
98
+ (ts * 2 + 4, 4, "shading"), (4, ts + 4, "depth"),
99
+ (ts + 4, ts + 4, "normal"),
100
+ (ts * 2 + 4, ts + 4, "recon")]:
101
+ draw.text((x, y), name, fill=(255, 255, 0))
102
+ return grid
model.py ADDED
@@ -0,0 +1,338 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """ShadeNet-3.2 generator (standalone, no Lightning dependency).
2
+
3
+ ParallelUNet v3: dual parallel encoders (vanilla UNet + frozen MobileNetV2),
4
+ depthwise-separable factorized convs, full H/32 bottleneck, and a patch
5
+ dictionary output tail -> 8ch intrinsic maps in [-1, 1]:
6
+ [0:3] albedo | [3:4] relative depth (0=near) | [4:7] normal | [7:8] shading
7
+ """
8
+ import math
9
+
10
+ import torch
11
+ import torch.nn as nn
12
+ import torchvision
13
+ from torchvision.models.feature_extraction import create_feature_extractor
14
+
15
+ OUT_CH = 8
16
+ WIDTH_MULT = 0.9
17
+
18
+
19
+ def _num_groups(ch: int) -> int:
20
+ g = min(32, ch)
21
+ while ch % g != 0:
22
+ g -= 1
23
+ return g
24
+
25
+
26
+ def _scale(width_mult):
27
+ """Channel scaler: width_mult rounded to multiples of 8 (GroupNorm-safe)."""
28
+ w = float(width_mult)
29
+
30
+ def C(n):
31
+ return max(8, int(round(n * w / 8.0) * 8))
32
+
33
+ return C
34
+
35
+
36
+ class ChannelLinear(nn.Module):
37
+ """Per-channel learnable affine: y = x * weight + bias."""
38
+
39
+ def __init__(self, channels: int, init_scale: float = 0.01):
40
+ super().__init__()
41
+ self.weight = nn.Parameter(
42
+ torch.empty(1, channels, 1, 1).uniform_(-init_scale, init_scale) + 1.0
43
+ )
44
+ self.bias = nn.Parameter(
45
+ torch.empty(1, channels, 1, 1).uniform_(-init_scale, init_scale)
46
+ )
47
+
48
+ def forward(self, x):
49
+ return x * self.weight + self.bias
50
+
51
+
52
+ class DoubleConv(nn.Module):
53
+ """Factorized (1,3)+(3,1) conv pair with GN+ELU, reflect padding."""
54
+
55
+ def __init__(self, in_ch, out_ch):
56
+ super().__init__()
57
+ g = _num_groups(out_ch)
58
+ self.conv = nn.Sequential(
59
+ nn.Conv2d(in_ch, out_ch, (1, 3), padding=(0, 1),
60
+ padding_mode="reflect", bias=False),
61
+ nn.GroupNorm(g, out_ch),
62
+ nn.ELU(inplace=True),
63
+ nn.Conv2d(out_ch, out_ch, (3, 1), padding=(1, 0),
64
+ padding_mode="reflect", bias=False),
65
+ nn.GroupNorm(g, out_ch),
66
+ nn.ELU(inplace=True),
67
+ )
68
+
69
+ def forward(self, x):
70
+ return self.conv(x)
71
+
72
+
73
+ class DSDoubleConv(nn.Module):
74
+ """Depthwise-separable DoubleConv: same geometry, ~3x fewer params."""
75
+
76
+ def __init__(self, in_ch, out_ch):
77
+ super().__init__()
78
+ g = _num_groups(out_ch)
79
+ self.conv = nn.Sequential(
80
+ nn.Conv2d(in_ch, in_ch, (1, 3), padding=(0, 1),
81
+ padding_mode="reflect", groups=in_ch, bias=False),
82
+ nn.Conv2d(in_ch, out_ch, 1, bias=False),
83
+ nn.GroupNorm(g, out_ch),
84
+ nn.ELU(inplace=True),
85
+ nn.Conv2d(out_ch, out_ch, (3, 1), padding=(1, 0),
86
+ padding_mode="reflect", groups=out_ch, bias=False),
87
+ nn.Conv2d(out_ch, out_ch, 1, bias=False),
88
+ nn.GroupNorm(g, out_ch),
89
+ nn.ELU(inplace=True),
90
+ )
91
+
92
+ def forward(self, x):
93
+ return self.conv(x)
94
+
95
+
96
+ class Down(nn.Module):
97
+ def __init__(self, in_ch, out_ch, block=DSDoubleConv):
98
+ super().__init__()
99
+ self.pool = nn.AvgPool2d(2)
100
+ self.conv = block(in_ch, out_ch)
101
+
102
+ def forward(self, x):
103
+ return self.conv(self.pool(x))
104
+
105
+
106
+ class Up(nn.Module):
107
+ def __init__(self, in_ch, out_ch, skip_ch=None, block=DSDoubleConv):
108
+ super().__init__()
109
+ skip_ch = skip_ch or in_ch
110
+ self.up = nn.ConvTranspose2d(in_ch, out_ch, 2, stride=2)
111
+ self.conv = block(out_ch + skip_ch, out_ch)
112
+
113
+ def forward(self, x, skip):
114
+ x = self.up(x)
115
+ return self.conv(torch.cat([skip, x], dim=1))
116
+
117
+
118
+ class PatchDictionaryBias(nn.Module):
119
+ """Soft patch-dictionary prior: tile -> softmax over atoms -> blend."""
120
+
121
+ def __init__(self, channels=OUT_CH, patch=16, n_atoms=1024, hidden=128,
122
+ deblock=True, blur_radius=2):
123
+ super().__init__()
124
+ self.patch = int(patch)
125
+ self.n_atoms = int(n_atoms)
126
+ self.atoms = nn.Parameter(
127
+ torch.zeros(n_atoms, channels, patch, patch))
128
+ self.desc = nn.Conv2d(channels, channels, 4, stride=4,
129
+ groups=channels, bias=False)
130
+ d = channels * (patch // 4) * (patch // 4)
131
+ self.addr = nn.Sequential(
132
+ nn.Linear(d, hidden),
133
+ nn.ELU(inplace=True),
134
+ nn.Linear(hidden, n_atoms),
135
+ )
136
+ self.blend = nn.Conv2d(2 * channels, channels, 1, bias=True)
137
+ self.blur_recon = bool(deblock)
138
+ self.blur_radius = int(blur_radius)
139
+ r = self.blur_radius
140
+ row = torch.tensor([math.comb(2 * r, i) for i in range(2 * r + 1)],
141
+ dtype=torch.float32)
142
+ row = row / row.sum()
143
+ k2 = (row[:, None] * row[None, :])[None, None].expand(
144
+ channels, 1, -1, -1).contiguous()
145
+ self.register_buffer("blur_k", k2)
146
+
147
+ def _gauss_blur(self, t):
148
+ r = self.blur_radius
149
+ t = torch.nn.functional.pad(t, (r, r, r, r), mode="reflect")
150
+ return torch.nn.functional.conv2d(t, self.blur_k, groups=t.shape[1])
151
+
152
+ def _deblock(self, recon, b, c, nh, nw, p):
153
+ img = (recon.reshape(b, nh, nw, c, p, p)
154
+ .permute(0, 3, 1, 4, 2, 5)
155
+ .reshape(b, c, nh * p, nw * p))
156
+ r = self.blur_radius
157
+ blurred = self._gauss_blur(img)
158
+ H, W = img.shape[-2:]
159
+ ih = torch.arange(H, device=img.device)
160
+ iw = torch.arange(W, device=img.device)
161
+ mh = ((ih % p) < r) | ((ih % p) >= p - r)
162
+ mw = ((iw % p) < r) | ((iw % p) >= p - r)
163
+ mask = (mh[:, None] | mw[None, :]).to(img.dtype)[None, None]
164
+ img = img + mask * (blurred - img)
165
+ return (img.reshape(b, c, nh, p, nw, p)
166
+ .permute(0, 2, 4, 1, 3, 5)
167
+ .reshape(b * nh * nw, c, p, p))
168
+
169
+ def forward(self, x):
170
+ b, c, h, w = x.shape
171
+ p = self.patch
172
+ nh, nw = h // p, w // p
173
+ tiles = (x.reshape(b, c, nh, p, nw, p)
174
+ .permute(0, 2, 4, 1, 3, 5)
175
+ .reshape(b * nh * nw, c, p, p))
176
+ feat = self.desc(tiles).flatten(1)
177
+ wts = torch.softmax(self.addr(feat), dim=1)
178
+ recon = torch.einsum("na,achw->nchw", wts, self.atoms)
179
+ if self.blur_recon:
180
+ recon = self._deblock(recon, b, c, nh, nw, p)
181
+ both = torch.cat([tiles, recon], dim=1)
182
+ out = self.blend(both).reshape(b, nh, nw, c, p, p)
183
+ return out.permute(0, 3, 1, 4, 2, 5).reshape(b, c, nh * p, nw * p)
184
+
185
+
186
+ class VanillaEncoder(nn.Module):
187
+ def __init__(self, in_ch=3, width_mult=1.0):
188
+ super().__init__()
189
+ C = _scale(width_mult)
190
+ self.inc = DoubleConv(in_ch, C(64))
191
+ self.down1 = Down(C(64), C(128))
192
+ self.down2 = Down(C(128), C(256))
193
+ self.down3 = Down(C(256), C(512))
194
+ self.down4 = Down(C(512), C(256))
195
+ self.down5 = Down(C(256), C(512))
196
+ self.mem1 = ChannelLinear(C(64))
197
+ self.mem2 = ChannelLinear(C(128))
198
+ self.mem3 = ChannelLinear(C(256))
199
+ self.mem4 = ChannelLinear(C(512))
200
+ self.mem5 = ChannelLinear(C(256))
201
+ self.mem6 = ChannelLinear(C(512))
202
+
203
+ def forward(self, x):
204
+ u1 = self.mem1(self.inc(x))
205
+ u2 = self.mem2(self.down1(u1))
206
+ u3 = self.mem3(self.down2(u2))
207
+ u4 = self.mem4(self.down3(u3))
208
+ u5 = self.mem5(self.down4(u4))
209
+ u6 = self.mem6(self.down5(u5))
210
+ return u1, u2, u3, u4, u5, u6
211
+
212
+
213
+ class MobileEncoder(nn.Module):
214
+ def __init__(self):
215
+ super().__init__()
216
+ mean = torch.tensor([0.485, 0.456, 0.406]).view(1, 3, 1, 1)
217
+ std = torch.tensor([0.229, 0.224, 0.225]).view(1, 3, 1, 1)
218
+ self.register_buffer("shift", 1.0 - 2.0 * mean)
219
+ self.register_buffer("scale", 1.0 / (2.0 * std))
220
+ mbnet = torchvision.models.mobilenet_v2(weights="IMAGENET1K_V1")
221
+ self.trunk = create_feature_extractor(
222
+ mbnet,
223
+ return_nodes={
224
+ "features.0": "m0", "features.2": "m1", "features.6": "m2",
225
+ "features.13": "m3", "features.17": "m4",
226
+ },
227
+ )
228
+ for p in self.trunk.parameters():
229
+ p.requires_grad = False
230
+
231
+ def forward(self, x):
232
+ feats = self.trunk((x + self.shift) * self.scale)
233
+ return feats["m0"], feats["m1"], feats["m2"], feats["m3"], feats["m4"]
234
+
235
+
236
+ class FusionBottleneck(nn.Module):
237
+ def __init__(self, width_mult=1.0):
238
+ super().__init__()
239
+ C = _scale(width_mult)
240
+ self.m_proj = nn.Conv2d(320, C(96), 1, bias=False)
241
+ self.fusion = DSDoubleConv(C(512) + C(96), C(512))
242
+ self.mem = ChannelLinear(C(512))
243
+
244
+ def forward(self, u6, m4):
245
+ return self.mem(self.fusion(torch.cat([u6, self.m_proj(m4)], dim=1)))
246
+
247
+
248
+ class FusedDecoder(nn.Module):
249
+ def __init__(self, width_mult=1.0):
250
+ super().__init__()
251
+ C = _scale(width_mult)
252
+ self.up0 = Up(C(512), C(256), skip_ch=C(256) + 96)
253
+ self.up1 = Up(C(256), C(256), skip_ch=C(512) + 32)
254
+ self.up2 = Up(C(256), C(256), skip_ch=C(256) + 24)
255
+ self.up3 = Up(C(256), C(128), skip_ch=C(128) + 32)
256
+ self.up4 = Up(C(128), C(64), skip_ch=C(64))
257
+ self.mem0 = ChannelLinear(C(256))
258
+ self.mem1 = ChannelLinear(C(256))
259
+ self.mem2 = ChannelLinear(C(256))
260
+ self.mem3 = ChannelLinear(C(128))
261
+ self.mem4 = ChannelLinear(C(64))
262
+
263
+ def forward(self, b, uni, mob):
264
+ u1, u2, u3, u4, u5 = uni
265
+ m0, m1, m2, m3 = mob
266
+ d0 = self.mem0(self.up0(b, torch.cat([u5, m3], dim=1)))
267
+ d1 = self.mem1(self.up1(d0, torch.cat([u4, m2], dim=1)))
268
+ d2 = self.mem2(self.up2(d1, torch.cat([u3, m1], dim=1)))
269
+ d3 = self.mem3(self.up3(d2, torch.cat([u2, m0], dim=1)))
270
+ return self.mem4(self.up4(d3, u1))
271
+
272
+
273
+ class IntrinsicHead(nn.Module):
274
+ def __init__(self, out_ch=OUT_CH, width_mult=1.0):
275
+ super().__init__()
276
+ C = _scale(width_mult)
277
+ self.head = DoubleConv(C(64) + 3, C(32))
278
+ self.head_out = nn.Conv2d(C(32), out_ch, 3, padding=1,
279
+ padding_mode="reflect")
280
+
281
+ def forward(self, d4, x):
282
+ return self.head_out(self.head(torch.cat([d4, x], dim=1)))
283
+
284
+
285
+ class DictTail(nn.Module):
286
+ """Patch-dictionary prior (blends internally, deblocked) -> tanh."""
287
+
288
+ def __init__(self, out_ch=OUT_CH, deblock=True, dict_atoms=1024):
289
+ super().__init__()
290
+ self.dict = PatchDictionaryBias(channels=out_ch, patch=16,
291
+ n_atoms=dict_atoms, deblock=deblock)
292
+
293
+ def forward(self, logits):
294
+ return torch.tanh(self.dict(logits))
295
+
296
+
297
+ class ParallelUNet(nn.Module):
298
+ def __init__(self, in_ch=3, out_ch=OUT_CH, width_mult=WIDTH_MULT,
299
+ bias_res=384, deblock=True, dict_atoms=1024):
300
+ super().__init__()
301
+ self.encoder = VanillaEncoder(in_ch, width_mult)
302
+ self.mobile = MobileEncoder()
303
+ self.bottleneck = FusionBottleneck(width_mult)
304
+ self.decoder = FusedDecoder(width_mult)
305
+ self.head = IntrinsicHead(out_ch, width_mult)
306
+ self.tail = DictTail(out_ch, deblock=deblock, dict_atoms=dict_atoms)
307
+
308
+ def forward(self, x):
309
+ u1, u2, u3, u4, u5, u6 = self.encoder(x)
310
+ m0, m1, m2, m3, m4 = self.mobile(x)
311
+ b = self.bottleneck(u6, m4)
312
+ d4 = self.decoder(b, (u1, u2, u3, u4, u5), (m0, m1, m2, m3))
313
+ return self.tail(self.head(d4, x))
314
+
315
+
316
+ def load_shadenet32(checkpoint_path, device="cpu", use_ema=True,
317
+ width_mult=WIDTH_MULT) -> ParallelUNet:
318
+ """Build the generator and load shadenet32.ckpt (Lightning or raw format).
319
+
320
+ Strips the `generator.` prefix from Lightning checkpoints and applies the
321
+ EMA shadow unless use_ema=False.
322
+ """
323
+ model = ParallelUNet(out_ch=OUT_CH, width_mult=width_mult)
324
+ ckpt = torch.load(checkpoint_path, map_location=device, weights_only=False)
325
+ sd = ckpt.get("state_dict", ckpt)
326
+ if any(k.startswith("generator.") for k in sd):
327
+ sd = {k[len("generator."):]: v for k, v in sd.items()
328
+ if k.startswith("generator.")}
329
+ model.load_state_dict(sd, strict=False)
330
+ if use_ema:
331
+ ema = ckpt.get("ema_generator") or {}
332
+ if ema:
333
+ model.load_state_dict(ema, strict=False)
334
+ print(f"Using EMA weights ({len(ema)} tensors).")
335
+ else:
336
+ print("No EMA shadow in checkpoint, using raw weights.")
337
+ model.eval().to(device)
338
+ return model
onnx/model.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:364b1676cdc6e3bab1d3035ea575f513a81cffe8ec34180509999d2ff46d4f9d
3
+ size 20053917
onnx/model_fp16.onnx ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5375b0c34610fb73f0df22b29e03dc3049f69827882f62f4e1fa74f7b40b2b93
3
+ size 10208498
requirements-space.txt ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ gradio
2
+ onnxruntime
3
+ pillow
4
+ numpy
5
+ spaces
requirements.txt ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ torch
2
+ torchvision
3
+ pillow
4
+ numpy
5
+ onnxruntime
6
+ gradio