diff --git a/.gitattributes b/.gitattributes index a44593ae5713e18791683b78f75492d9e11143d6..5974fbd2e3025bdb6e92f1751651493b47475001 100644 --- a/.gitattributes +++ b/.gitattributes @@ -36,3 +36,5 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text migrator/checkpoints/hub/facebookresearch_dinov2_main/docs/Cell-DINO.png filter=lfs diff=lfs merge=lfs -text migrator/checkpoints/hub/facebookresearch_dinov2_main/docs/ChannelAdaptiveDINO.png filter=lfs diff=lfs merge=lfs -text migrator/code/mv-sam3d-for-6d-v2/mvsam3d/data/__pycache__/adapters.cpython-311.pyc filter=lfs diff=lfs merge=lfs -text +migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/adapters.cpython-311.pyc filter=lfs diff=lfs merge=lfs -text +migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slatflow_dataset.cpython-311.pyc filter=lfs diff=lfs merge=lfs -text diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow.yaml.bak_predatveiw b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow.yaml.bak_predatveiw new file mode 100644 index 0000000000000000000000000000000000000000..ff42f9d0e5b45c9ccc060a1a610df85ca0637b14 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow.yaml.bak_predatveiw @@ -0,0 +1,169 @@ +# Training config for the multi-view SLAT flow (SlatFlowModel, SLAT_GEN_PLAN). +# ALL training hyperparameters live here; train_slat_flow.py reads them and +# nothing training-related is hardcoded in Python. Launch: +# SPARSE_ATTN_BACKEND=sdpa PYTHONPATH=. \ +# python mvsam3d/train/train_slat_flow.py --config configs/train_slatflow.yaml +# (multi-GPU: torchrun --nproc_per_node=N ...) +# Per-key overrides: --set key.subkey=value (repeatable). + +# --------------------------------------------------------------------------- # +# data -> TrellisSlatFlowDataset (DATA_FORMAT.md layout + per-view RGB assets). +# Smoke: toys4k200 objects with the toys4k1k inputs/renders fallback for +# RGB+bbox (V=4 views: front/side/oside/back — camera-verified identical to the +# stored views.npz). Production roots get views/rgb + views/bbox.npy + +# slat/slat_sam3d.npz per SLAT_GEN_PLAN §8 (precompute jobs pending). +# --------------------------------------------------------------------------- # +data: + roots: + - /lp-dev/jonghoon/mv-sam3d-6d-code/smoke_data/toys4k200 + # Explicit validation roots (same policy as train_ssflow.yaml): non-empty -> + # val = ALL objects under val_roots, train = roots minus those ids. + # NOTE: val_omni3d has no toys4k1k RGB fallback assets yet, so the smoke val + # set is toys4k-only until the production rgb/bbox precompute lands. + val_roots: + - /lp-dev/jonghoon/mv-sam3d-6d-code/smoke_data/val_toys4k100 + # --- view-subset augmentation (the ONLY train-time stochasticity besides + # the fresh invisible-row noise in x0) --- + # smoke data has 4 views/object; production renders 24 (paper §3.1.1). + # TRELLIS-style per-object voxel FILTER (structured_latent.py + # filter_metadata): objects with more active 64^3 voxels are DROPPED at + # index-build time (never downsampled). 20000 = the single-object SLAT + # OOM-safe bound (PLAN/BATCH_SIZE_BENCH.md; drops ~4.8% of trellis500k). + # 0 disables the filter. + max_num_voxels: 20000 + min_views: 1 # |S| ~ Uniform{min_views..max_views} per sample + max_views: 4 # clamped per object to its available view count + val_num_views: 2 # deterministic view count for validation samples + aug_seed: 0 # seeds the per-(epoch,idx) view-subset RNG + # --- legacy hash split (used only when val_roots is empty) --- + val_fraction: 0.05 + split_seed: 0 + +# --------------------------------------------------------------------------- # +# model -> SlatFlowModel. The generator stack (FlowMatching -> CFG -> +# SLatFlowModelTdfyWrapper) is hydra-built from the vendored slat_generator.yaml +# with the pipeline's effective inference config (steps=25, cfg_strength=1, +# cfg_interval=[0,500], rescale_t=1) baked in by build_slat_generator. +# --------------------------------------------------------------------------- # +model: + # Warm-start from the deployed SLAT generator (526/526 tensors, verified). + pretrained_ckpt: ${MIGRATOR_CHECKPOINTS}/hf/slat_generator.ckpt + # Local soft-mask cross-attention bias (§4c). Parity gating (§10 step 4) + # PASSED 2026-09-01 (wrapped/zero-bias bit-identical to the pristine stack on + # 3 toys4k objects), so the bias is ENABLED for training. Set false to + # reproduce the gating configuration. + enable_soft_mask: true # local -10 soft-mask in cross-attention + enable_plucker: true # Plücker ray embedding on non-anchor views + bias_value: -10.0 # fixed, fp16-safe; NEVER -inf (§4c / risk #7) + # Backbone param dtype: false = lossless fp16->fp32 upcast for training + # (optimizer-friendly); true = yaml-verbatim fp16 torso (the parity config). + fp16_torso: false + # Gradient checkpointing on the 24 backbone DiT cross blocks (per-block + # .use_checkpoint flip, same toggle the batch-size bench used): recompute + # activations in backward -> big memory cut, ~modest step-time cost. + # false = current behavior (no checkpointing). + use_checkpoint: false + # Train the pretrained EmbedderFuser projections + slot embeddings (the DINO + # towers inside are always frozen). false freezes the whole fuser so only + # the backbone + Plücker FC train. + train_condition_embedder: true + +# --------------------------------------------------------------------------- # +# optim + train loop +# --------------------------------------------------------------------------- # +optim: + lr: 3.0e-5 + weight_decay: 0.0 # AdamW + grad_clip: 1.0 # max grad norm + +# --------------------------------------------------------------------------- # +# ema — exponential moving average of the TRAINABLE weights (fp32 shadow, +# updated after every optimizer step; frozen DINO towers excluded). +# Disabled by default = existing behavior unchanged. State is saved as a +# per-checkpoint sidecar (step_XXXX_ema.pt, next to the _optimizer.pt sidecar) +# and restored automatically on resume_from. +# --------------------------------------------------------------------------- # +ema: + enable: true + rate: 0.999 # ema = rate*ema + (1-rate)*param + eval_with_ema: true # swap EMA weights in for the in-loop validation + +train: + # Effective batch = batch_objects x grad_accum_steps x world_size objects + # per optimizer step. + # batch_objects = 1 (default): loader bs=1 + collate_single, one object per + # backbone forward (batching purely via grad_accum_steps). + # batch_objects > 1: TRUE multi-object batching (collate_batched + + # SlatFlowModel.flow_step_batched) — B objects concatenated into ONE + # sparse coord set with the batch-index column, ONE backbone forward, + # independent t per object. Requires grad_accum_steps == 1. + steps: 20000 + num_workers: 16 # CPU work: crops + projections + Plücker per view + batch_objects: 1 # objects per GPU per forward (true batching if >1) + grad_accum_steps: 8 + # --- LR schedule (reused fork build_scheduler: warmup -> cosine -> min_lr) --- + scheduler: cosine # cosine | constant + warmup_steps: 500 + min_lr: 1.0e-6 + # --- resume --- + resume_from: "" # checkpoint path to resume (empty = fresh start) + log_every: 20 + ckpt_every: 5000 + ckpt_dir: ${MIGRATOR_CACHE}/ckpt_slatflow + amp: true + amp_dtype: bf16 # bf16 | fp16 (fp16 enables GradScaler) + seed: 0 + # --- distributed (DDP via torchrun; world_size=1 transparently skips it) --- + backend: ddp # ddp | deepspeed + deepspeed_config: "" + # --- validation: held-out flow v-MSE only (fixed RNG). Decode + + # faithfulness eval = the TrainTest phase (frozen-pipeline decode). --- + val_every: 5000 # optimizer steps between validation passes (0 = never) + val_max_batches: 8 # val objects per rank per pass + +# --------------------------------------------------------------------------- # +# wandb — rank-0 experiment logging (train/loss|lr|grad_norm per log_every step, +# val/* per validation pass). enabled: false makes every wandb call a no-op. +# Extra metric dicts can be pushed from anywhere via +# from mvsam3d.train.train_slat_flow import log_metrics +# log_metrics({"val/chamfer": ..., "val/psnr": ...}, step) +# --------------------------------------------------------------------------- # +wandb: + enabled: true + project: "mvsam3d-slatflow" + entity: "alphabet1" + name: null # null -> W&B autogenerates the run name + dir: "/lp-dev/jonghoon/mv-mesh/wandb" + +# --------------------------------------------------------------------------- # +# val_appforce — APPEARANCE-FORCING validation (mvsam3d/train/val_appforce.py): +# the trained flow through the exact reported protocol (batch_appforce_sam3d +# sam3d_geom_dino 2v -> evaluate_appforce.py) on toys4k-100 (textured) + +# OmniObject3D-117, every `every` optimizer steps (+ step 0 baseline), EMA +# weights, ALL ranks sample their shard then run the reference-env decode + +# eval on their own GPU; rank 0 merges and logs val_af///. +# Stage-1 (frozen z1fwd geometry forcing) is cached ONCE by +# tools/val_stage1_ref.py --mode stage1 --exp --views 2 --out +# //2v +# --------------------------------------------------------------------------- # +val_appforce: + enabled: true + every: 5000 + at_step0: true + views: 2 + schedules: [seed, reinject] # seed = the training x0 construction; reinject = the reported schedule + seed: 42 + limit: 0 # debug: cap objects per dataset + batch_size: 6 # objects per BATCHED 25-step ODE solve (1 = old single path) + single: false # A/B: force the one-object-per-solve path + decode_procs: 2 # concurrent reference-env decode shards per rank + ref_threads: 8 # OMP/MKL cap for every reference-env subprocess + stage1_cache: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/stage1 + out_root: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/runs + ref_python: /lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/bin/python + eval_dataset_flag: toys4k # evaluate_appforce --dataset (both sets use the toys4k path, as reported) + datasets: + - name: toys4k100_tex + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/toys4k100_tex + - name: omni3d + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/omni3d diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_prod.yaml b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_prod.yaml index 6d845774911283090104f457336233eaf572b13e..02c8a22c9234ad3f0f2bd33bc111a35ca709b637 100644 --- a/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_prod.yaml +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_prod.yaml @@ -63,20 +63,20 @@ data: # Non-empty -> TarSlatFlowDataset reads latents/cond/cameras straight from # the .tar shards (GL->CV flip + alpha-bbox crop); `roots` is ignored. --- tar_roots: [] - # --- condition variants for the .tar path (2026-09-24; all default OFF) --- + # --- condition sources for the .tar path (2026-09-24) — ON for production --- # use_recgen_images: index recgen//{cameras.json,NN.jpg,NN_mask.png} - # (RecGen's own background image + mask + rigid cube cameras) from any tar - # under tar_roots, joined to latents/.npz by id; latent-only slats WITH - # RecGen images become trainable (else they are skipped). - # p_recgen_image: P(RecGen image instead of the white cond render) when an - # object has both; single-source objects always use what they have. + # (RecGen's own background image + visible mask + rigid cube cameras) from + # ANY tar under tar_roots, joined to latents/.npz by the same id string. + # ONE POOL: an object with both sources draws its anchor from the union of + # its white cond views and RecGen views; the anchor's source decides ALL aux + # views (never mixed). RecGen anchors go through the same deployed + # preprocess_slat_image recipe into item["image"] (background kept, alpha = + # mask). Latent-only RecGen slats become trainable; white-only objects are + # drawn exactly as before. # cond_cameras_fallback: white-cond objects without cameras//cond_cameras.json # (24-view slat5/slat5rs batches, 22,755 objs) read cond//transforms.json. - # For the Hub flat layout (slat_train_flat/) the intended setting is - # use_recgen_images: true, p_recgen_image: 0.5, cond_cameras_fallback: true - use_recgen_images: false - p_recgen_image: 0.0 - cond_cameras_fallback: false + use_recgen_images: true + cond_cameras_fallback: true # --------------------------------------------------------------------------- # # model -> SlatFlowModel. The generator stack (FlowMatching -> CFG -> diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_prod.yaml.bak_predatveiw b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_prod.yaml.bak_predatveiw new file mode 100644 index 0000000000000000000000000000000000000000..1651e65aed8cd0028e70e2827f66ffa94be66b90 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_prod.yaml.bak_predatveiw @@ -0,0 +1,177 @@ +# Training config for the multi-view SLAT flow (SlatFlowModel, SLAT_GEN_PLAN). +# ALL training hyperparameters live here; train_slat_flow.py reads them and +# nothing training-related is hardcoded in Python. Launch: +# SPARSE_ATTN_BACKEND=sdpa PYTHONPATH=. \ +# python mvsam3d/train/train_slat_flow.py --config configs/train_slatflow.yaml +# (multi-GPU: torchrun --nproc_per_node=N ...) +# Per-key overrides: --set key.subkey=value (repeatable). + +# --------------------------------------------------------------------------- # +# data -> TrellisSlatFlowDataset (DATA_FORMAT.md layout + per-view RGB assets). +# Smoke: toys4k200 objects with the toys4k1k inputs/renders fallback for +# RGB+bbox (V=4 views: front/side/oside/back — camera-verified identical to the +# stored views.npz). Production roots get views/rgb + views/bbox.npy + +# slat/slat_sam3d.npz per SLAT_GEN_PLAN §8 (precompute jobs pending). +# --------------------------------------------------------------------------- # +data: + roots: # merged production set (remote 29.4k + local, DATA_FORMAT via tools/adapt_slat50k.py) + - /data/mv_mesh_data/slat_train/dataset + # Explicit validation roots (same policy as train_ssflow.yaml): non-empty -> + # val = ALL objects under val_roots, train = roots minus those ids. + # NOTE: val_omni3d has no toys4k1k RGB fallback assets yet, so the smoke val + # set is toys4k-only until the production rgb/bbox precompute lands. + val_roots: [] # smoke val set has NO official x1 -> flow-loss val would crash; use a hash split of the production roots instead + # --- view-subset augmentation (the ONLY train-time stochasticity besides + # the fresh invisible-row noise in x0) --- + # smoke data has 4 views/object; production renders 24 (paper §3.1.1). + # TRELLIS-style per-object voxel FILTER (structured_latent.py + # filter_metadata): objects with more active 64^3 voxels are DROPPED at + # index-build time (never downsampled). 20000 = the single-object SLAT + # OOM-safe bound (PLAN/BATCH_SIZE_BENCH.md; drops ~4.8% of trellis500k). + # 0 disables the filter. + max_num_voxels: 20000 + min_views: 1 # |S| ~ Uniform{min_views..max_views} per sample + max_views: 4 # clamped per object to its available view count + val_num_views: 2 # deterministic view count for validation samples + aug_seed: 0 # seeds the per-(epoch,idx) view-subset RNG + # --- legacy hash split (used only when val_roots is empty) --- + val_fraction: 0.005 # ~185 held-out production objects for the flow-loss val + split_seed: 0 + # --- SS-stage seed MIXTURE (unified none/single/multi ratio) --- + # p_seed_none = P(EMPTY appearance seed -> x0 all-noise, no forcing) + # p_seed_single = P(single-view seed -> n_sub == 1); remainder = multi. + # BOTH 0.0 (default) -> LEGACY n_sub ~ U{min_views..max_views}, unchanged. + p_seed_none: 0.0 + p_seed_single: 0.0 + # --- RecGen loader (additive / opt-in): dirs of released-SLAT objects + # (slat_coords.npy + slat_feats.npy + cond_image_*/cond_mask_* + camera + # metadata), fed to SLAT via matting. Empty = disabled. CAVEAT: run a + # GPU decode-compat test before training on RecGen's RELEASED latents. --- + recgen_roots: [] + # encoded_store recgen root with canonical cube cameras (pose_cube.npz / + # views.npz) looked up by sha; needed for CORRECT recgen voxel + # visibility (raw view_metadata alone is geometrically approximate). + recgen_pose_store: null + +# --------------------------------------------------------------------------- # +# model -> SlatFlowModel. The generator stack (FlowMatching -> CFG -> +# SLatFlowModelTdfyWrapper) is hydra-built from the vendored slat_generator.yaml +# with the pipeline's effective inference config (steps=25, cfg_strength=1, +# cfg_interval=[0,500], rescale_t=1) baked in by build_slat_generator. +# --------------------------------------------------------------------------- # +model: + # Warm-start from the deployed SLAT generator (526/526 tensors, verified). + pretrained_ckpt: ${MIGRATOR_CHECKPOINTS}/hf/slat_generator.ckpt + # Local soft-mask cross-attention bias (§4c). Parity gating (§10 step 4) + # PASSED 2026-09-01 (wrapped/zero-bias bit-identical to the pristine stack on + # 3 toys4k objects), so the bias is ENABLED for training. Set false to + # reproduce the gating configuration. + enable_soft_mask: true # local -10 soft-mask in cross-attention + enable_plucker: true # Plücker ray embedding on non-anchor views + bias_value: -10.0 # fixed, fp16-safe; NEVER -inf (§4c / risk #7) + # Backbone param dtype: false = lossless fp16->fp32 upcast for training + # (optimizer-friendly); true = yaml-verbatim fp16 torso (the parity config). + fp16_torso: false + # Gradient checkpointing on the 24 backbone DiT cross blocks (per-block + # .use_checkpoint flip, same toggle the batch-size bench used): recompute + # activations in backward -> big memory cut, ~modest step-time cost. + # false = current behavior (no checkpointing). + use_checkpoint: false # user choice 2026-09-03: +20% speed; worst case (3x20k-voxel objects) peaks ~69 GB, watchdog resumes on OOM + # Train the pretrained EmbedderFuser projections + slot embeddings (the DINO + # towers inside are always frozen). false freezes the whole fuser so only + # the backbone + Plücker FC train. + train_condition_embedder: true + +# --------------------------------------------------------------------------- # +# optim + train loop +# --------------------------------------------------------------------------- # +optim: + lr: 3.0e-5 + weight_decay: 0.0 # AdamW + grad_clip: 1.0 # max grad norm + +# --------------------------------------------------------------------------- # +# ema — exponential moving average of the TRAINABLE weights (fp32 shadow, +# updated after every optimizer step; frozen DINO towers excluded). +# Disabled by default = existing behavior unchanged. State is saved as a +# per-checkpoint sidecar (step_XXXX_ema.pt, next to the _optimizer.pt sidecar) +# and restored automatically on resume_from. +# --------------------------------------------------------------------------- # +ema: + enable: true + rate: 0.999 # ema = rate*ema + (1-rate)*param + eval_with_ema: true # swap EMA weights in for the in-loop validation + +train: + # Loader batch size is PINNED to 1 (per-object coord sets, collate_single); + # the effective batch is grad_accum_steps x world_size objects per step. + steps: 100000 # ~32 epochs of 37k objects at 12 objects/step + num_workers: 16 # CPU work: crops + projections + Plücker per view + grad_accum_steps: 1 # NO accumulation: effective batch = batch_objects x world_size = 12 + batch_objects: 3 # objects per GPU per optimizer step (true batched flow step) + # --- LR schedule (reused fork build_scheduler: warmup -> cosine -> min_lr) --- + scheduler: cosine # cosine | constant + warmup_steps: 500 + min_lr: 1.0e-6 + # --- resume --- + resume_from: "" # checkpoint path to resume (empty = fresh start) + log_every: 20 + ckpt_every: 2500 + ckpt_dir: /data/mv_mesh_data/ckpt/slatflow_prod_20260902 # /data (1.1T free); 40 ckpts x 5 GB model+ema, optimizer sidecars pruned to the last 2 + amp: true + amp_dtype: bf16 # bf16 | fp16 (fp16 enables GradScaler) + seed: 0 + # --- distributed (DDP via torchrun; world_size=1 transparently skips it) --- + backend: ddp # ddp | deepspeed + deepspeed_config: "" + # --- validation: held-out flow v-MSE only (fixed RNG). Decode + + # faithfulness eval = the TrainTest phase (frozen-pipeline decode). --- + val_every: 5000 # optimizer steps between validation passes (0 = never) + val_max_batches: 8 # val objects per rank per pass + +# --------------------------------------------------------------------------- # +# wandb — rank-0 experiment logging (train/loss|lr|grad_norm per log_every step, +# val/* per validation pass). enabled: false makes every wandb call a no-op. +# Extra metric dicts can be pushed from anywhere via +# from mvsam3d.train.train_slat_flow import log_metrics +# log_metrics({"val/chamfer": ..., "val/psnr": ...}, step) +# --------------------------------------------------------------------------- # +wandb: + enabled: true + project: "mvsam3d-slatflow" + entity: "alphabet1" + name: "slatflow_prod_bs12_lr3e-5_ema0.999_100k" + dir: "/lp-dev/jonghoon/mv-mesh/wandb" + +# --------------------------------------------------------------------------- # +# val_appforce — APPEARANCE-FORCING validation (mvsam3d/train/val_appforce.py): +# the trained flow through the exact reported protocol (batch_appforce_sam3d +# sam3d_geom_dino 2v -> evaluate_appforce.py) on toys4k-100 (textured) + +# OmniObject3D-117, every `every` optimizer steps (+ step 0 baseline), EMA +# weights, ALL ranks sample their shard then run the reference-env decode + +# eval on their own GPU; rank 0 merges and logs val_af///. +# Stage-1 (frozen z1fwd geometry forcing) is cached ONCE by +# tools/val_stage1_ref.py --mode stage1 --exp --views 2 --out +# //2v +# --------------------------------------------------------------------------- # +val_appforce: + enabled: false # in-loop val OFF: tools/val_daemon.py validates every 5k ckpt on GPUs 2-3 (training never pauses) + every: 5000 + at_step0: false # first appearance-forcing val at 5000 (step-0 baseline = the parity run) + views: 2 + schedules: [seed, reinject] # seed = the training x0 construction; reinject = the reported schedule + seed: 42 + limit: 0 # debug: cap objects per dataset + batch_size: 6 # objects per BATCHED 25-step ODE solve (1 = old single path) + single: true # batched B=6 measured 0.8x slower than single; parity verified but no gain + decode_procs: 1 # 2 procs x 23 GB OOMed rank 3 at the 10k val + ref_threads: 8 # OMP/MKL cap for every reference-env subprocess + stage1_cache: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/stage1 + out_root: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/runs + ref_python: /lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/bin/python + eval_dataset_flag: toys4k # evaluate_appforce --dataset (both sets use the toys4k path, as reported) + datasets: + - name: toys4k100_tex + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/toys4k100_tex + - name: omni3d + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/omni3d diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume20k_2gpu.yaml.bak_predatveiw b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume20k_2gpu.yaml.bak_predatveiw new file mode 100644 index 0000000000000000000000000000000000000000..5529795912e6a3e0fa9865ce67a14323237f9842 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume20k_2gpu.yaml.bak_predatveiw @@ -0,0 +1,177 @@ +# Training config for the multi-view SLAT flow (SlatFlowModel, SLAT_GEN_PLAN). +# ALL training hyperparameters live here; train_slat_flow.py reads them and +# nothing training-related is hardcoded in Python. Launch: +# SPARSE_ATTN_BACKEND=sdpa PYTHONPATH=. \ +# python mvsam3d/train/train_slat_flow.py --config configs/train_slatflow.yaml +# (multi-GPU: torchrun --nproc_per_node=N ...) +# Per-key overrides: --set key.subkey=value (repeatable). + +# --------------------------------------------------------------------------- # +# data -> TrellisSlatFlowDataset (DATA_FORMAT.md layout + per-view RGB assets). +# Smoke: toys4k200 objects with the toys4k1k inputs/renders fallback for +# RGB+bbox (V=4 views: front/side/oside/back — camera-verified identical to the +# stored views.npz). Production roots get views/rgb + views/bbox.npy + +# slat/slat_sam3d.npz per SLAT_GEN_PLAN §8 (precompute jobs pending). +# --------------------------------------------------------------------------- # +data: + roots: # POOL for the 20k->30k resume (user 2026-09-07): objects present at the 20k stop + 5,000 randomly sampled NEW slats (local+HF); lists in .debug/slat_prod/pool_{base,new5k}_shas.txt + - /data/mv_mesh_data/slat_train/dataset_pool_20k_plus5k + # Explicit validation roots (same policy as train_ssflow.yaml): non-empty -> + # val = ALL objects under val_roots, train = roots minus those ids. + # NOTE: val_omni3d has no toys4k1k RGB fallback assets yet, so the smoke val + # set is toys4k-only until the production rgb/bbox precompute lands. + val_roots: [] # smoke val set has NO official x1 -> flow-loss val would crash; use a hash split of the production roots instead + # --- view-subset augmentation (the ONLY train-time stochasticity besides + # the fresh invisible-row noise in x0) --- + # smoke data has 4 views/object; production renders 24 (paper §3.1.1). + # TRELLIS-style per-object voxel FILTER (structured_latent.py + # filter_metadata): objects with more active 64^3 voxels are DROPPED at + # index-build time (never downsampled). 20000 = the single-object SLAT + # OOM-safe bound (PLAN/BATCH_SIZE_BENCH.md; drops ~4.8% of trellis500k). + # 0 disables the filter. + max_num_voxels: 20000 + min_views: 1 # |S| ~ Uniform{min_views..max_views} per sample + max_views: 4 # clamped per object to its available view count + val_num_views: 2 # deterministic view count for validation samples + aug_seed: 0 # seeds the per-(epoch,idx) view-subset RNG + # --- legacy hash split (used only when val_roots is empty) --- + val_fraction: 0.005 # ~185 held-out production objects for the flow-loss val + split_seed: 0 + # --- SS-stage seed MIXTURE (unified none/single/multi ratio) --- + # p_seed_none = P(EMPTY appearance seed -> x0 all-noise, no forcing) + # p_seed_single = P(single-view seed -> n_sub == 1); remainder = multi. + # BOTH 0.0 (default) -> LEGACY n_sub ~ U{min_views..max_views}, unchanged. + p_seed_none: 0.0 + p_seed_single: 0.0 + # --- RecGen loader (additive / opt-in): dirs of released-SLAT objects + # (slat_coords.npy + slat_feats.npy + cond_image_*/cond_mask_* + camera + # metadata), fed to SLAT via matting. Empty = disabled. CAVEAT: run a + # GPU decode-compat test before training on RecGen's RELEASED latents. --- + recgen_roots: [] + # encoded_store recgen root with canonical cube cameras (pose_cube.npz / + # views.npz) looked up by sha; needed for CORRECT recgen voxel + # visibility (raw view_metadata alone is geometrically approximate). + recgen_pose_store: null + +# --------------------------------------------------------------------------- # +# model -> SlatFlowModel. The generator stack (FlowMatching -> CFG -> +# SLatFlowModelTdfyWrapper) is hydra-built from the vendored slat_generator.yaml +# with the pipeline's effective inference config (steps=25, cfg_strength=1, +# cfg_interval=[0,500], rescale_t=1) baked in by build_slat_generator. +# --------------------------------------------------------------------------- # +model: + # Warm-start from the deployed SLAT generator (526/526 tensors, verified). + pretrained_ckpt: ${MIGRATOR_CHECKPOINTS}/hf/slat_generator.ckpt + # Local soft-mask cross-attention bias (§4c). Parity gating (§10 step 4) + # PASSED 2026-09-01 (wrapped/zero-bias bit-identical to the pristine stack on + # 3 toys4k objects), so the bias is ENABLED for training. Set false to + # reproduce the gating configuration. + enable_soft_mask: true # local -10 soft-mask in cross-attention + enable_plucker: true # Plücker ray embedding on non-anchor views + bias_value: -10.0 # fixed, fp16-safe; NEVER -inf (§4c / risk #7) + # Backbone param dtype: false = lossless fp16->fp32 upcast for training + # (optimizer-friendly); true = yaml-verbatim fp16 torso (the parity config). + fp16_torso: false + # Gradient checkpointing on the 24 backbone DiT cross blocks (per-block + # .use_checkpoint flip, same toggle the batch-size bench used): recompute + # activations in backward -> big memory cut, ~modest step-time cost. + # false = current behavior (no checkpointing). + use_checkpoint: true # 2-GPU resume run (user 2026-09-07): 6 objects/GPU needs activation checkpointing + # Train the pretrained EmbedderFuser projections + slot embeddings (the DINO + # towers inside are always frozen). false freezes the whole fuser so only + # the backbone + Plücker FC train. + train_condition_embedder: true + +# --------------------------------------------------------------------------- # +# optim + train loop +# --------------------------------------------------------------------------- # +optim: + lr: 3.0e-5 + weight_decay: 0.0 # AdamW + grad_clip: 1.0 # max grad norm + +# --------------------------------------------------------------------------- # +# ema — exponential moving average of the TRAINABLE weights (fp32 shadow, +# updated after every optimizer step; frozen DINO towers excluded). +# Disabled by default = existing behavior unchanged. State is saved as a +# per-checkpoint sidecar (step_XXXX_ema.pt, next to the _optimizer.pt sidecar) +# and restored automatically on resume_from. +# --------------------------------------------------------------------------- # +ema: + enable: true + rate: 0.999 # ema = rate*ema + (1-rate)*param + eval_with_ema: true # swap EMA weights in for the in-loop validation + +train: + # Loader batch size is PINNED to 1 (per-object coord sets, collate_single); + # the effective batch is grad_accum_steps x world_size objects per step. + steps: 100000 # ~32 epochs of 37k objects at 12 objects/step + num_workers: 16 # CPU work: crops + projections + Plücker per view + grad_accum_steps: 1 # NO accumulation: effective batch = batch_objects x world_size = 12 + batch_objects: 6 # 6 objects/GPU x 2 GPUs = effective batch 12 (same as the 4-GPU run) + # --- LR schedule (reused fork build_scheduler: warmup -> cosine -> min_lr) --- + scheduler: cosine # cosine | constant + warmup_steps: 500 + min_lr: 1.0e-6 + # --- resume --- + resume_from: "/data/mv_mesh_data/ckpt/slatflow_prod_20260902/step_0020000.pt" # resume the 20k prod ckpt (+ema/optimizer sidecars) + log_every: 20 + ckpt_every: 2500 + ckpt_dir: /data/mv_mesh_data/ckpt/slatflow_resume20k_2gpu + amp: true + amp_dtype: bf16 # bf16 | fp16 (fp16 enables GradScaler) + seed: 0 + # --- distributed (DDP via torchrun; world_size=1 transparently skips it) --- + backend: ddp # ddp | deepspeed + deepspeed_config: "" + # --- validation: held-out flow v-MSE only (fixed RNG). Decode + + # faithfulness eval = the TrainTest phase (frozen-pipeline decode). --- + val_every: 5000 # optimizer steps between validation passes (0 = never) + val_max_batches: 8 # val objects per rank per pass + +# --------------------------------------------------------------------------- # +# wandb — rank-0 experiment logging (train/loss|lr|grad_norm per log_every step, +# val/* per validation pass). enabled: false makes every wandb call a no-op. +# Extra metric dicts can be pushed from anywhere via +# from mvsam3d.train.train_slat_flow import log_metrics +# log_metrics({"val/chamfer": ..., "val/psnr": ...}, step) +# --------------------------------------------------------------------------- # +wandb: + enabled: true + project: "mvsam3d-slatflow" + entity: "alphabet1" + name: "slatflow_resume20k_2gpu_bs12_pool55k" + dir: "/lp-dev/jonghoon/mv-mesh/wandb" + +# --------------------------------------------------------------------------- # +# val_appforce — APPEARANCE-FORCING validation (mvsam3d/train/val_appforce.py): +# the trained flow through the exact reported protocol (batch_appforce_sam3d +# sam3d_geom_dino 2v -> evaluate_appforce.py) on toys4k-100 (textured) + +# OmniObject3D-117, every `every` optimizer steps (+ step 0 baseline), EMA +# weights, ALL ranks sample their shard then run the reference-env decode + +# eval on their own GPU; rank 0 merges and logs val_af///. +# Stage-1 (frozen z1fwd geometry forcing) is cached ONCE by +# tools/val_stage1_ref.py --mode stage1 --exp --views 2 --out +# //2v +# --------------------------------------------------------------------------- # +val_appforce: + enabled: false # in-loop val OFF: tools/val_daemon.py validates every 5k ckpt on GPUs 2-3 (training never pauses) + every: 5000 + at_step0: false # first appearance-forcing val at 5000 (step-0 baseline = the parity run) + views: 2 + schedules: [seed, reinject] # seed = the training x0 construction; reinject = the reported schedule + seed: 42 + limit: 0 # debug: cap objects per dataset + batch_size: 6 # objects per BATCHED 25-step ODE solve (1 = old single path) + single: true # batched B=6 measured 0.8x slower than single; parity verified but no gain + decode_procs: 1 # 2 procs x 23 GB OOMed rank 3 at the 10k val + ref_threads: 8 # OMP/MKL cap for every reference-env subprocess + stage1_cache: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/stage1 + out_root: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/runs + ref_python: /lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/bin/python + eval_dataset_flag: toys4k # evaluate_appforce --dataset (both sets use the toys4k path, as reported) + datasets: + - name: toys4k100_tex + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/toys4k100_tex + - name: omni3d + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/omni3d diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume20k_pslaug_2gpu.yaml.bak_predatveiw b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume20k_pslaug_2gpu.yaml.bak_predatveiw new file mode 100644 index 0000000000000000000000000000000000000000..60cbaabc4531c5c216fbbf29b3c92d7b31852cfc --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume20k_pslaug_2gpu.yaml.bak_predatveiw @@ -0,0 +1,186 @@ +# Training config for the multi-view SLAT flow (SlatFlowModel, SLAT_GEN_PLAN). +# ALL training hyperparameters live here; train_slat_flow.py reads them and +# nothing training-related is hardcoded in Python. Launch: +# SPARSE_ATTN_BACKEND=sdpa PYTHONPATH=. \ +# python mvsam3d/train/train_slat_flow.py --config configs/train_slatflow.yaml +# (multi-GPU: torchrun --nproc_per_node=N ...) +# Per-key overrides: --set key.subkey=value (repeatable). + +# --------------------------------------------------------------------------- # +# data -> TrellisSlatFlowDataset (DATA_FORMAT.md layout + per-view RGB assets). +# Smoke: toys4k200 objects with the toys4k1k inputs/renders fallback for +# RGB+bbox (V=4 views: front/side/oside/back — camera-verified identical to the +# stored views.npz). Production roots get views/rgb + views/bbox.npy + +# slat/slat_sam3d.npz per SLAT_GEN_PLAN §8 (precompute jobs pending). +# --------------------------------------------------------------------------- # +data: + roots: # POOL for the 20k->30k PSL-AUG resume (user 2026-09-07): + # (a) the 50,621 base-pool objects present at the 20k stop + # (.debug/slat_prod/pool_base_shas.txt), and + # (b) ALL new PSL-augmentation objects from + # /data/mv_mesh_data/psl_aug/objects (psl_aug_shas.txt). + # The pool root is symlink-only, built by + # /lp-dev/jonghoon/mv-sam3d-6d-code/build_pool_base_pslaug.sh + # as /{remote,local,pslaug}/objects//. + # psl_aug objects carry NO special weighting: scan_objects + # flattens all three sources into one uniformly-sampled index. + - /data/mv_mesh_data/slat_train/dataset_pool_base_pslaug + # Explicit validation roots (same policy as train_ssflow.yaml): non-empty -> + # val = ALL objects under val_roots, train = roots minus those ids. + # NOTE: val_omni3d has no toys4k1k RGB fallback assets yet, so the smoke val + # set is toys4k-only until the production rgb/bbox precompute lands. + val_roots: [] # smoke val set has NO official x1 -> flow-loss val would crash; use a hash split of the production roots instead + # --- view-subset augmentation (the ONLY train-time stochasticity besides + # the fresh invisible-row noise in x0) --- + # smoke data has 4 views/object; production renders 24 (paper §3.1.1). + # TRELLIS-style per-object voxel FILTER (structured_latent.py + # filter_metadata): objects with more active 64^3 voxels are DROPPED at + # index-build time (never downsampled). 20000 = the single-object SLAT + # OOM-safe bound (PLAN/BATCH_SIZE_BENCH.md; drops ~4.8% of trellis500k). + # 0 disables the filter. + max_num_voxels: 20000 + min_views: 1 # |S| ~ Uniform{min_views..max_views} per sample + max_views: 4 # clamped per object to its available view count + val_num_views: 2 # deterministic view count for validation samples + aug_seed: 0 # seeds the per-(epoch,idx) view-subset RNG + # --- legacy hash split (used only when val_roots is empty) --- + val_fraction: 0.005 # ~185 held-out production objects for the flow-loss val + split_seed: 0 + # --- SS-stage seed MIXTURE (unified none/single/multi ratio) --- + # p_seed_none = P(EMPTY appearance seed -> x0 all-noise, no forcing) + # p_seed_single = P(single-view seed -> n_sub == 1); remainder = multi. + # BOTH 0.0 (default) -> LEGACY n_sub ~ U{min_views..max_views}, unchanged. + p_seed_none: 0.0 + p_seed_single: 0.0 + # --- RecGen loader (additive / opt-in): dirs of released-SLAT objects + # (slat_coords.npy + slat_feats.npy + cond_image_*/cond_mask_* + camera + # metadata), fed to SLAT via matting. Empty = disabled. CAVEAT: run a + # GPU decode-compat test before training on RecGen's RELEASED latents. --- + recgen_roots: [] + # encoded_store recgen root with canonical cube cameras (pose_cube.npz / + # views.npz) looked up by sha; needed for CORRECT recgen voxel + # visibility (raw view_metadata alone is geometrically approximate). + recgen_pose_store: null + +# --------------------------------------------------------------------------- # +# model -> SlatFlowModel. The generator stack (FlowMatching -> CFG -> +# SLatFlowModelTdfyWrapper) is hydra-built from the vendored slat_generator.yaml +# with the pipeline's effective inference config (steps=25, cfg_strength=1, +# cfg_interval=[0,500], rescale_t=1) baked in by build_slat_generator. +# --------------------------------------------------------------------------- # +model: + # Warm-start from the deployed SLAT generator (526/526 tensors, verified). + pretrained_ckpt: ${MIGRATOR_CHECKPOINTS}/hf/slat_generator.ckpt + # Local soft-mask cross-attention bias (§4c). Parity gating (§10 step 4) + # PASSED 2026-09-01 (wrapped/zero-bias bit-identical to the pristine stack on + # 3 toys4k objects), so the bias is ENABLED for training. Set false to + # reproduce the gating configuration. + enable_soft_mask: true # local -10 soft-mask in cross-attention + enable_plucker: true # Plücker ray embedding on non-anchor views + bias_value: -10.0 # fixed, fp16-safe; NEVER -inf (§4c / risk #7) + # Backbone param dtype: false = lossless fp16->fp32 upcast for training + # (optimizer-friendly); true = yaml-verbatim fp16 torso (the parity config). + fp16_torso: false + # Gradient checkpointing on the 24 backbone DiT cross blocks (per-block + # .use_checkpoint flip, same toggle the batch-size bench used): recompute + # activations in backward -> big memory cut, ~modest step-time cost. + # false = current behavior (no checkpointing). + use_checkpoint: true # 2-GPU resume run (user 2026-09-07): 6 objects/GPU needs activation checkpointing + # Train the pretrained EmbedderFuser projections + slot embeddings (the DINO + # towers inside are always frozen). false freezes the whole fuser so only + # the backbone + Plücker FC train. + train_condition_embedder: true + +# --------------------------------------------------------------------------- # +# optim + train loop +# --------------------------------------------------------------------------- # +optim: + lr: 3.0e-5 + weight_decay: 0.0 # AdamW + grad_clip: 1.0 # max grad norm + +# --------------------------------------------------------------------------- # +# ema — exponential moving average of the TRAINABLE weights (fp32 shadow, +# updated after every optimizer step; frozen DINO towers excluded). +# Disabled by default = existing behavior unchanged. State is saved as a +# per-checkpoint sidecar (step_XXXX_ema.pt, next to the _optimizer.pt sidecar) +# and restored automatically on resume_from. +# --------------------------------------------------------------------------- # +ema: + enable: true + rate: 0.999 # ema = rate*ema + (1-rate)*param + eval_with_ema: true # swap EMA weights in for the in-loop validation + +train: + # Loader batch size is PINNED to 1 (per-object coord sets, collate_single); + # the effective batch is grad_accum_steps x world_size objects per step. + steps: 100000 # ~32 epochs of 37k objects at 12 objects/step + num_workers: 16 # CPU work: crops + projections + Plücker per view + grad_accum_steps: 1 # NO accumulation: effective batch = batch_objects x world_size = 12 + batch_objects: 6 # 6 objects/GPU x 2 GPUs = effective batch 12 (same as the 4-GPU run) + # --- LR schedule (reused fork build_scheduler: warmup -> cosine -> min_lr) --- + scheduler: cosine # cosine | constant + warmup_steps: 500 + min_lr: 1.0e-6 + # --- resume --- + resume_from: "/data/mv_mesh_data/ckpt/slatflow_prod_20260902/step_0020000.pt" # resume the 20k prod ckpt (+ema/optimizer sidecars) + log_every: 20 + ckpt_every: 2500 + ckpt_dir: /data/mv_mesh_data/ckpt/slatflow_resume20k_pslaug_2gpu + amp: true + amp_dtype: bf16 # bf16 | fp16 (fp16 enables GradScaler) + seed: 0 + # --- distributed (DDP via torchrun; world_size=1 transparently skips it) --- + backend: ddp # ddp | deepspeed + deepspeed_config: "" + # --- validation: held-out flow v-MSE only (fixed RNG). Decode + + # faithfulness eval = the TrainTest phase (frozen-pipeline decode). --- + val_every: 5000 # optimizer steps between validation passes (0 = never) + val_max_batches: 8 # val objects per rank per pass + +# --------------------------------------------------------------------------- # +# wandb — rank-0 experiment logging (train/loss|lr|grad_norm per log_every step, +# val/* per validation pass). enabled: false makes every wandb call a no-op. +# Extra metric dicts can be pushed from anywhere via +# from mvsam3d.train.train_slat_flow import log_metrics +# log_metrics({"val/chamfer": ..., "val/psnr": ...}, step) +# --------------------------------------------------------------------------- # +wandb: + enabled: true + project: "mvsam3d-slatflow" + entity: "alphabet1" + name: "slatflow_resume20k_pslaug_bs12" + dir: "/lp-dev/jonghoon/mv-mesh/wandb" + +# --------------------------------------------------------------------------- # +# val_appforce — APPEARANCE-FORCING validation (mvsam3d/train/val_appforce.py): +# the trained flow through the exact reported protocol (batch_appforce_sam3d +# sam3d_geom_dino 2v -> evaluate_appforce.py) on toys4k-100 (textured) + +# OmniObject3D-117, every `every` optimizer steps (+ step 0 baseline), EMA +# weights, ALL ranks sample their shard then run the reference-env decode + +# eval on their own GPU; rank 0 merges and logs val_af///. +# Stage-1 (frozen z1fwd geometry forcing) is cached ONCE by +# tools/val_stage1_ref.py --mode stage1 --exp --views 2 --out +# //2v +# --------------------------------------------------------------------------- # +val_appforce: + enabled: false # in-loop val OFF: tools/val_daemon.py validates every 5k ckpt on GPUs 2-3 (training never pauses) + every: 5000 + at_step0: false # first appearance-forcing val at 5000 (step-0 baseline = the parity run) + views: 2 + schedules: [seed, reinject] # seed = the training x0 construction; reinject = the reported schedule + seed: 42 + limit: 0 # debug: cap objects per dataset + batch_size: 6 # objects per BATCHED 25-step ODE solve (1 = old single path) + single: true # batched B=6 measured 0.8x slower than single; parity verified but no gain + decode_procs: 1 # 2 procs x 23 GB OOMed rank 3 at the 10k val + ref_threads: 8 # OMP/MKL cap for every reference-env subprocess + stage1_cache: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/stage1 + out_root: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/runs + ref_python: /lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/bin/python + eval_dataset_flag: toys4k # evaluate_appforce --dataset (both sets use the toys4k path, as reported) + datasets: + - name: toys4k100_tex + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/toys4k100_tex + - name: omni3d + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/omni3d diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume22p5k_pslfinal_2gpu.yaml.bak_predatveiw b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume22p5k_pslfinal_2gpu.yaml.bak_predatveiw new file mode 100644 index 0000000000000000000000000000000000000000..df529cb820f3bfeba23d8a46b6b0f3381da95fa2 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume22p5k_pslfinal_2gpu.yaml.bak_predatveiw @@ -0,0 +1,188 @@ +# Training config for the multi-view SLAT flow (SlatFlowModel, SLAT_GEN_PLAN). +# ALL training hyperparameters live here; train_slat_flow.py reads them and +# nothing training-related is hardcoded in Python. Launch: +# SPARSE_ATTN_BACKEND=sdpa PYTHONPATH=. \ +# python mvsam3d/train/train_slat_flow.py --config configs/train_slatflow.yaml +# (multi-GPU: torchrun --nproc_per_node=N ...) +# Per-key overrides: --set key.subkey=value (repeatable). + +# --------------------------------------------------------------------------- # +# data -> TrellisSlatFlowDataset (DATA_FORMAT.md layout + per-view RGB assets). +# Smoke: toys4k200 objects with the toys4k1k inputs/renders fallback for +# RGB+bbox (V=4 views: front/side/oside/back — camera-verified identical to the +# stored views.npz). Production roots get views/rgb + views/bbox.npy + +# slat/slat_sam3d.npz per SLAT_GEN_PLAN §8 (precompute jobs pending). +# --------------------------------------------------------------------------- # +data: + roots: # POOL for the 22.5k->30k PSL-FINAL resume (user 2026-09-08): + # (a) the 50,621 base-pool objects present at the 20k stop + # (.debug/slat_prod/pool_base_shas.txt), and + # (b) the MERGED PSL set listed in + # /data/mv_mesh_data/psl_aug_occ/psl_final_shas.txt + # ('\t': occ -> psl_aug_occ/objects, the + # occlusion-heavy replacement set; aug -> psl_aug/objects, + # the earlier whole-view set). + # Symlink-only root built by + # /lp-dev/jonghoon/mv-sam3d-6d-code/build_pool_base_pslfinal.sh + # as /{remote,local,pslocc,pslaug}/objects//. + # No special weighting: one uniformly-sampled index. + - /data/mv_mesh_data/slat_train/dataset_pool_base_pslfinal + # Explicit validation roots (same policy as train_ssflow.yaml): non-empty -> + # val = ALL objects under val_roots, train = roots minus those ids. + # NOTE: val_omni3d has no toys4k1k RGB fallback assets yet, so the smoke val + # set is toys4k-only until the production rgb/bbox precompute lands. + val_roots: [] # smoke val set has NO official x1 -> flow-loss val would crash; use a hash split of the production roots instead + # --- view-subset augmentation (the ONLY train-time stochasticity besides + # the fresh invisible-row noise in x0) --- + # smoke data has 4 views/object; production renders 24 (paper §3.1.1). + # TRELLIS-style per-object voxel FILTER (structured_latent.py + # filter_metadata): objects with more active 64^3 voxels are DROPPED at + # index-build time (never downsampled). 20000 = the single-object SLAT + # OOM-safe bound (PLAN/BATCH_SIZE_BENCH.md; drops ~4.8% of trellis500k). + # 0 disables the filter. + max_num_voxels: 20000 + min_views: 1 # |S| ~ Uniform{min_views..max_views} per sample + max_views: 4 # clamped per object to its available view count + val_num_views: 2 # deterministic view count for validation samples + aug_seed: 0 # seeds the per-(epoch,idx) view-subset RNG + # --- legacy hash split (used only when val_roots is empty) --- + val_fraction: 0.005 # ~185 held-out production objects for the flow-loss val + split_seed: 0 + # --- SS-stage seed MIXTURE (unified none/single/multi ratio) --- + # p_seed_none = P(EMPTY appearance seed -> x0 all-noise, no forcing) + # p_seed_single = P(single-view seed -> n_sub == 1); remainder = multi. + # BOTH 0.0 (default) -> LEGACY n_sub ~ U{min_views..max_views}, unchanged. + p_seed_none: 0.0 + p_seed_single: 0.0 + # --- RecGen loader (additive / opt-in): dirs of released-SLAT objects + # (slat_coords.npy + slat_feats.npy + cond_image_*/cond_mask_* + camera + # metadata), fed to SLAT via matting. Empty = disabled. CAVEAT: run a + # GPU decode-compat test before training on RecGen's RELEASED latents. --- + recgen_roots: [] + # encoded_store recgen root with canonical cube cameras (pose_cube.npz / + # views.npz) looked up by sha; needed for CORRECT recgen voxel + # visibility (raw view_metadata alone is geometrically approximate). + recgen_pose_store: null + +# --------------------------------------------------------------------------- # +# model -> SlatFlowModel. The generator stack (FlowMatching -> CFG -> +# SLatFlowModelTdfyWrapper) is hydra-built from the vendored slat_generator.yaml +# with the pipeline's effective inference config (steps=25, cfg_strength=1, +# cfg_interval=[0,500], rescale_t=1) baked in by build_slat_generator. +# --------------------------------------------------------------------------- # +model: + # Warm-start from the deployed SLAT generator (526/526 tensors, verified). + pretrained_ckpt: ${MIGRATOR_CHECKPOINTS}/hf/slat_generator.ckpt + # Local soft-mask cross-attention bias (§4c). Parity gating (§10 step 4) + # PASSED 2026-09-01 (wrapped/zero-bias bit-identical to the pristine stack on + # 3 toys4k objects), so the bias is ENABLED for training. Set false to + # reproduce the gating configuration. + enable_soft_mask: true # local -10 soft-mask in cross-attention + enable_plucker: true # Plücker ray embedding on non-anchor views + bias_value: -10.0 # fixed, fp16-safe; NEVER -inf (§4c / risk #7) + # Backbone param dtype: false = lossless fp16->fp32 upcast for training + # (optimizer-friendly); true = yaml-verbatim fp16 torso (the parity config). + fp16_torso: false + # Gradient checkpointing on the 24 backbone DiT cross blocks (per-block + # .use_checkpoint flip, same toggle the batch-size bench used): recompute + # activations in backward -> big memory cut, ~modest step-time cost. + # false = current behavior (no checkpointing). + use_checkpoint: true # 2-GPU resume run (user 2026-09-07): 6 objects/GPU needs activation checkpointing + # Train the pretrained EmbedderFuser projections + slot embeddings (the DINO + # towers inside are always frozen). false freezes the whole fuser so only + # the backbone + Plücker FC train. + train_condition_embedder: true + +# --------------------------------------------------------------------------- # +# optim + train loop +# --------------------------------------------------------------------------- # +optim: + lr: 3.0e-5 + weight_decay: 0.0 # AdamW + grad_clip: 1.0 # max grad norm + +# --------------------------------------------------------------------------- # +# ema — exponential moving average of the TRAINABLE weights (fp32 shadow, +# updated after every optimizer step; frozen DINO towers excluded). +# Disabled by default = existing behavior unchanged. State is saved as a +# per-checkpoint sidecar (step_XXXX_ema.pt, next to the _optimizer.pt sidecar) +# and restored automatically on resume_from. +# --------------------------------------------------------------------------- # +ema: + enable: true + rate: 0.999 # ema = rate*ema + (1-rate)*param + eval_with_ema: true # swap EMA weights in for the in-loop validation + +train: + # Loader batch size is PINNED to 1 (per-object coord sets, collate_single); + # the effective batch is grad_accum_steps x world_size objects per step. + steps: 100000 # ~32 epochs of 37k objects at 12 objects/step + num_workers: 16 # CPU work: crops + projections + Plücker per view + grad_accum_steps: 1 # NO accumulation: effective batch = batch_objects x world_size = 12 + batch_objects: 6 # 6 objects/GPU x 2 GPUs = effective batch 12 (same as the 4-GPU run) + # --- LR schedule (reused fork build_scheduler: warmup -> cosine -> min_lr) --- + scheduler: cosine # cosine | constant + warmup_steps: 500 + min_lr: 1.0e-6 + # --- resume --- + resume_from: "/data/mv_mesh_data/ckpt/slatflow_resume20k_pslaug_2gpu/step_0022500.pt" # resume the 22.5k pslaug ckpt (+ema/optimizer sidecars); pool swapped to base + merged PSL final set + log_every: 20 + ckpt_every: 2500 + ckpt_dir: /data/mv_mesh_data/ckpt/slatflow_resume22p5k_pslfinal_2gpu + amp: true + amp_dtype: bf16 # bf16 | fp16 (fp16 enables GradScaler) + seed: 0 + # --- distributed (DDP via torchrun; world_size=1 transparently skips it) --- + backend: ddp # ddp | deepspeed + deepspeed_config: "" + # --- validation: held-out flow v-MSE only (fixed RNG). Decode + + # faithfulness eval = the TrainTest phase (frozen-pipeline decode). --- + val_every: 5000 # optimizer steps between validation passes (0 = never) + val_max_batches: 8 # val objects per rank per pass + +# --------------------------------------------------------------------------- # +# wandb — rank-0 experiment logging (train/loss|lr|grad_norm per log_every step, +# val/* per validation pass). enabled: false makes every wandb call a no-op. +# Extra metric dicts can be pushed from anywhere via +# from mvsam3d.train.train_slat_flow import log_metrics +# log_metrics({"val/chamfer": ..., "val/psnr": ...}, step) +# --------------------------------------------------------------------------- # +wandb: + enabled: true + project: "mvsam3d-slatflow" + entity: "alphabet1" + name: "slatflow_resume22p5k_pslfinal_bs12" + dir: "/lp-dev/jonghoon/mv-mesh/wandb" + +# --------------------------------------------------------------------------- # +# val_appforce — APPEARANCE-FORCING validation (mvsam3d/train/val_appforce.py): +# the trained flow through the exact reported protocol (batch_appforce_sam3d +# sam3d_geom_dino 2v -> evaluate_appforce.py) on toys4k-100 (textured) + +# OmniObject3D-117, every `every` optimizer steps (+ step 0 baseline), EMA +# weights, ALL ranks sample their shard then run the reference-env decode + +# eval on their own GPU; rank 0 merges and logs val_af///. +# Stage-1 (frozen z1fwd geometry forcing) is cached ONCE by +# tools/val_stage1_ref.py --mode stage1 --exp --views 2 --out +# //2v +# --------------------------------------------------------------------------- # +val_appforce: + enabled: false # in-loop val OFF: tools/val_daemon.py validates every 5k ckpt on GPUs 2-3 (training never pauses) + every: 5000 + at_step0: false # first appearance-forcing val at 5000 (step-0 baseline = the parity run) + views: 2 + schedules: [seed, reinject] # seed = the training x0 construction; reinject = the reported schedule + seed: 42 + limit: 0 # debug: cap objects per dataset + batch_size: 6 # objects per BATCHED 25-step ODE solve (1 = old single path) + single: true # batched B=6 measured 0.8x slower than single; parity verified but no gain + decode_procs: 1 # 2 procs x 23 GB OOMed rank 3 at the 10k val + ref_threads: 8 # OMP/MKL cap for every reference-env subprocess + stage1_cache: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/stage1 + out_root: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/runs + ref_python: /lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/bin/python + eval_dataset_flag: toys4k # evaluate_appforce --dataset (both sets use the toys4k path, as reported) + datasets: + - name: toys4k100_tex + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/toys4k100_tex + - name: omni3d + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/omni3d diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume22p5k_pslocc_2gpu.yaml.bak_predatveiw b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume22p5k_pslocc_2gpu.yaml.bak_predatveiw new file mode 100644 index 0000000000000000000000000000000000000000..8130e4a7a3ccbfb8bf9b80834322191848aa632a --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/configs/train_slatflow_resume22p5k_pslocc_2gpu.yaml.bak_predatveiw @@ -0,0 +1,187 @@ +# Training config for the multi-view SLAT flow (SlatFlowModel, SLAT_GEN_PLAN). +# ALL training hyperparameters live here; train_slat_flow.py reads them and +# nothing training-related is hardcoded in Python. Launch: +# SPARSE_ATTN_BACKEND=sdpa PYTHONPATH=. \ +# python mvsam3d/train/train_slat_flow.py --config configs/train_slatflow.yaml +# (multi-GPU: torchrun --nproc_per_node=N ...) +# Per-key overrides: --set key.subkey=value (repeatable). + +# --------------------------------------------------------------------------- # +# data -> TrellisSlatFlowDataset (DATA_FORMAT.md layout + per-view RGB assets). +# Smoke: toys4k200 objects with the toys4k1k inputs/renders fallback for +# RGB+bbox (V=4 views: front/side/oside/back — camera-verified identical to the +# stored views.npz). Production roots get views/rgb + views/bbox.npy + +# slat/slat_sam3d.npz per SLAT_GEN_PLAN §8 (precompute jobs pending). +# --------------------------------------------------------------------------- # +data: + roots: # POOL for the 22.5k->30k PSL-OCCLUSION resume (user 2026-09-08): + # (a) the 50,621 base-pool objects present at the 20k stop + # (.debug/slat_prod/pool_base_shas.txt), and + # (b) ALL objects of the REPLACEMENT occlusion set + # /data/mv_mesh_data/psl_aug_occ/objects (3-4 whole + + # 4-5 occluded views per sample; psl_aug_occ_shas.txt). + # The earlier /data/mv_mesh_data/psl_aug set is NOT included. + # Symlink-only root built by + # /lp-dev/jonghoon/mv-sam3d-6d-code/build_pool_base_pslocc.sh + # as /{remote,local,pslocc}/objects//. + # No special weighting: one uniformly-sampled index. + - /data/mv_mesh_data/slat_train/dataset_pool_base_pslocc + # Explicit validation roots (same policy as train_ssflow.yaml): non-empty -> + # val = ALL objects under val_roots, train = roots minus those ids. + # NOTE: val_omni3d has no toys4k1k RGB fallback assets yet, so the smoke val + # set is toys4k-only until the production rgb/bbox precompute lands. + val_roots: [] # smoke val set has NO official x1 -> flow-loss val would crash; use a hash split of the production roots instead + # --- view-subset augmentation (the ONLY train-time stochasticity besides + # the fresh invisible-row noise in x0) --- + # smoke data has 4 views/object; production renders 24 (paper §3.1.1). + # TRELLIS-style per-object voxel FILTER (structured_latent.py + # filter_metadata): objects with more active 64^3 voxels are DROPPED at + # index-build time (never downsampled). 20000 = the single-object SLAT + # OOM-safe bound (PLAN/BATCH_SIZE_BENCH.md; drops ~4.8% of trellis500k). + # 0 disables the filter. + max_num_voxels: 20000 + min_views: 1 # |S| ~ Uniform{min_views..max_views} per sample + max_views: 4 # clamped per object to its available view count + val_num_views: 2 # deterministic view count for validation samples + aug_seed: 0 # seeds the per-(epoch,idx) view-subset RNG + # --- legacy hash split (used only when val_roots is empty) --- + val_fraction: 0.005 # ~185 held-out production objects for the flow-loss val + split_seed: 0 + # --- SS-stage seed MIXTURE (unified none/single/multi ratio) --- + # p_seed_none = P(EMPTY appearance seed -> x0 all-noise, no forcing) + # p_seed_single = P(single-view seed -> n_sub == 1); remainder = multi. + # BOTH 0.0 (default) -> LEGACY n_sub ~ U{min_views..max_views}, unchanged. + p_seed_none: 0.0 + p_seed_single: 0.0 + # --- RecGen loader (additive / opt-in): dirs of released-SLAT objects + # (slat_coords.npy + slat_feats.npy + cond_image_*/cond_mask_* + camera + # metadata), fed to SLAT via matting. Empty = disabled. CAVEAT: run a + # GPU decode-compat test before training on RecGen's RELEASED latents. --- + recgen_roots: [] + # encoded_store recgen root with canonical cube cameras (pose_cube.npz / + # views.npz) looked up by sha; needed for CORRECT recgen voxel + # visibility (raw view_metadata alone is geometrically approximate). + recgen_pose_store: null + +# --------------------------------------------------------------------------- # +# model -> SlatFlowModel. The generator stack (FlowMatching -> CFG -> +# SLatFlowModelTdfyWrapper) is hydra-built from the vendored slat_generator.yaml +# with the pipeline's effective inference config (steps=25, cfg_strength=1, +# cfg_interval=[0,500], rescale_t=1) baked in by build_slat_generator. +# --------------------------------------------------------------------------- # +model: + # Warm-start from the deployed SLAT generator (526/526 tensors, verified). + pretrained_ckpt: ${MIGRATOR_CHECKPOINTS}/hf/slat_generator.ckpt + # Local soft-mask cross-attention bias (§4c). Parity gating (§10 step 4) + # PASSED 2026-09-01 (wrapped/zero-bias bit-identical to the pristine stack on + # 3 toys4k objects), so the bias is ENABLED for training. Set false to + # reproduce the gating configuration. + enable_soft_mask: true # local -10 soft-mask in cross-attention + enable_plucker: true # Plücker ray embedding on non-anchor views + bias_value: -10.0 # fixed, fp16-safe; NEVER -inf (§4c / risk #7) + # Backbone param dtype: false = lossless fp16->fp32 upcast for training + # (optimizer-friendly); true = yaml-verbatim fp16 torso (the parity config). + fp16_torso: false + # Gradient checkpointing on the 24 backbone DiT cross blocks (per-block + # .use_checkpoint flip, same toggle the batch-size bench used): recompute + # activations in backward -> big memory cut, ~modest step-time cost. + # false = current behavior (no checkpointing). + use_checkpoint: true # 2-GPU resume run (user 2026-09-07): 6 objects/GPU needs activation checkpointing + # Train the pretrained EmbedderFuser projections + slot embeddings (the DINO + # towers inside are always frozen). false freezes the whole fuser so only + # the backbone + Plücker FC train. + train_condition_embedder: true + +# --------------------------------------------------------------------------- # +# optim + train loop +# --------------------------------------------------------------------------- # +optim: + lr: 3.0e-5 + weight_decay: 0.0 # AdamW + grad_clip: 1.0 # max grad norm + +# --------------------------------------------------------------------------- # +# ema — exponential moving average of the TRAINABLE weights (fp32 shadow, +# updated after every optimizer step; frozen DINO towers excluded). +# Disabled by default = existing behavior unchanged. State is saved as a +# per-checkpoint sidecar (step_XXXX_ema.pt, next to the _optimizer.pt sidecar) +# and restored automatically on resume_from. +# --------------------------------------------------------------------------- # +ema: + enable: true + rate: 0.999 # ema = rate*ema + (1-rate)*param + eval_with_ema: true # swap EMA weights in for the in-loop validation + +train: + # Loader batch size is PINNED to 1 (per-object coord sets, collate_single); + # the effective batch is grad_accum_steps x world_size objects per step. + steps: 100000 # ~32 epochs of 37k objects at 12 objects/step + num_workers: 16 # CPU work: crops + projections + Plücker per view + grad_accum_steps: 1 # NO accumulation: effective batch = batch_objects x world_size = 12 + batch_objects: 6 # 6 objects/GPU x 2 GPUs = effective batch 12 (same as the 4-GPU run) + # --- LR schedule (reused fork build_scheduler: warmup -> cosine -> min_lr) --- + scheduler: cosine # cosine | constant + warmup_steps: 500 + min_lr: 1.0e-6 + # --- resume --- + resume_from: "/data/mv_mesh_data/ckpt/slatflow_resume20k_pslaug_2gpu/step_0022500.pt" # resume the 22.5k pslaug ckpt (+ema/optimizer sidecars); pool swapped to base+psl_aug_occ + log_every: 20 + ckpt_every: 2500 + ckpt_dir: /data/mv_mesh_data/ckpt/slatflow_resume22p5k_pslocc_2gpu + amp: true + amp_dtype: bf16 # bf16 | fp16 (fp16 enables GradScaler) + seed: 0 + # --- distributed (DDP via torchrun; world_size=1 transparently skips it) --- + backend: ddp # ddp | deepspeed + deepspeed_config: "" + # --- validation: held-out flow v-MSE only (fixed RNG). Decode + + # faithfulness eval = the TrainTest phase (frozen-pipeline decode). --- + val_every: 5000 # optimizer steps between validation passes (0 = never) + val_max_batches: 8 # val objects per rank per pass + +# --------------------------------------------------------------------------- # +# wandb — rank-0 experiment logging (train/loss|lr|grad_norm per log_every step, +# val/* per validation pass). enabled: false makes every wandb call a no-op. +# Extra metric dicts can be pushed from anywhere via +# from mvsam3d.train.train_slat_flow import log_metrics +# log_metrics({"val/chamfer": ..., "val/psnr": ...}, step) +# --------------------------------------------------------------------------- # +wandb: + enabled: true + project: "mvsam3d-slatflow" + entity: "alphabet1" + name: "slatflow_resume22p5k_pslocc_bs12" + dir: "/lp-dev/jonghoon/mv-mesh/wandb" + +# --------------------------------------------------------------------------- # +# val_appforce — APPEARANCE-FORCING validation (mvsam3d/train/val_appforce.py): +# the trained flow through the exact reported protocol (batch_appforce_sam3d +# sam3d_geom_dino 2v -> evaluate_appforce.py) on toys4k-100 (textured) + +# OmniObject3D-117, every `every` optimizer steps (+ step 0 baseline), EMA +# weights, ALL ranks sample their shard then run the reference-env decode + +# eval on their own GPU; rank 0 merges and logs val_af///. +# Stage-1 (frozen z1fwd geometry forcing) is cached ONCE by +# tools/val_stage1_ref.py --mode stage1 --exp --views 2 --out +# //2v +# --------------------------------------------------------------------------- # +val_appforce: + enabled: false # in-loop val OFF: tools/val_daemon.py validates every 5k ckpt on GPUs 2-3 (training never pauses) + every: 5000 + at_step0: false # first appearance-forcing val at 5000 (step-0 baseline = the parity run) + views: 2 + schedules: [seed, reinject] # seed = the training x0 construction; reinject = the reported schedule + seed: 42 + limit: 0 # debug: cap objects per dataset + batch_size: 6 # objects per BATCHED 25-step ODE solve (1 = old single path) + single: true # batched B=6 measured 0.8x slower than single; parity verified but no gain + decode_procs: 1 # 2 procs x 23 GB OOMed rank 3 at the 10k val + ref_threads: 8 # OMP/MKL cap for every reference-env subprocess + stage1_cache: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/stage1 + out_root: /lp-dev/jonghoon/mv-sam3d-6d-code/val_appforce/runs + ref_python: /lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/bin/python + eval_dataset_flag: toys4k # evaluate_appforce --dataset (both sets use the toys4k path, as reported) + datasets: + - name: toys4k100_tex + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/toys4k100_tex + - name: omni3d + exp: /lp-dev/jonghoon/mv-mesh/exp_faithfulness/omni3d diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/bunny_003.pt b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/bunny_003.pt new file mode 100644 index 0000000000000000000000000000000000000000..fe5d5d168ce6438a6e9cc22b11b365979babe464 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/bunny_003.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:14502f014a45dee714dbd919e394b419a72e95419897303bc7386eed8e0a5508 +size 20996178 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/bunny_008.pt b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/bunny_008.pt new file mode 100644 index 0000000000000000000000000000000000000000..660af17266cfbe0dd826aac7ce2778b44490a98e --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/bunny_008.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92e4ff59ac962b00dc5a5975913c5e952ece70ab3a640d25c0727bd68265ae08 +size 20511442 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/chair_026.pt b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/chair_026.pt new file mode 100644 index 0000000000000000000000000000000000000000..bf508f1b8bac54743652d0a3b90623da9a8c7920 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/chair_026.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bdf5dc60ce5b61d76bb78762bd5895cdb21ff3a49252f63d9ef8007c73ed3ffd +size 20690002 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/ours_both.log b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/ours_both.log new file mode 100644 index 0000000000000000000000000000000000000000..44eceeba238c47717dd0b23cbf95c17066f86dec --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/ours_both.log @@ -0,0 +1,51 @@ +/lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/lib/python3.11/site-packages/torch/cuda/__init__.py:61: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you. + import pynvml # type: ignore[import] +2026-09-01 01:49:09.196 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.sparse:__from_env:39 - [SPARSE] Backend: spconv, Attention: sdpa +2026-09-01 01:49:12.938 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.attention:__from_env:30 - [ATTENTION] Using backend: sdpa +[SPARSE][CONV] spconv algo: native +[slat-pretrained] slat_generator: loaded 526/526 own tensors from '_base_models.generator.*' (0 missing, 0 unexpected) +2026-09-01 01:49:25.245 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:32 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-09-01 01:49:29.957 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:51 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-09-01 01:49:29.961 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:32 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-09-01 01:49:33.784 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:51 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +[slat-pretrained] slat_condition_embedder: loaded 699/699 own tensors from '_base_models.condition_embedder.*' (0 missing, 0 unexpected) +[slat-pretrained] slat_generator: loaded 526/526 own tensors from '_base_models.generator.*' (0 missing, 0 unexpected) +2026-09-01 01:49:44.041 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:32 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-09-01 01:49:47.428 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:51 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-09-01 01:49:47.431 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:32 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-09-01 01:49:51.389 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:51 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +[slat-pretrained] slat_condition_embedder: loaded 699/699 own tensors from '_base_models.condition_embedder.*' (0 missing, 0 unexpected) +[ours] bunny_003 (N=7984, ref steps=25 strength=1.0 rescale_t=1.0 | deployed self-noise floor=0.000e+00 tol=5.000e-02) + [raw] + cond BIT-IDENTICAL shape=(1, 5496, 1024) dtype=torch.float16 max|d|=0.000e+00 + slat_feats_noise WITHIN-FLOOR max|d|=6.896e-03 (tol 5.000e-02) + slat_feats_forced WITHIN-FLOOR max|d|=9.228e-03 (tol 5.000e-02) + [wrapped/zero-bias] + cond BIT-IDENTICAL shape=(1, 5496, 1024) dtype=torch.float16 max|d|=0.000e+00 + slat_feats_noise WITHIN-FLOOR max|d|=6.896e-03 (tol 5.000e-02) + wrapped==raw (noise) BIT-IDENTICAL shape=(7984, 8) dtype=torch.float32 max|d|=0.000e+00 + slat_feats_forced WITHIN-FLOOR max|d|=9.228e-03 (tol 5.000e-02) + wrapped==raw (forced) BIT-IDENTICAL shape=(7984, 8) dtype=torch.float32 max|d|=0.000e+00 +[ours] bunny_008 (N=4619, ref steps=25 strength=1.0 rescale_t=1.0 | deployed self-noise floor=0.000e+00 tol=5.000e-02) + [raw] + cond BIT-IDENTICAL shape=(1, 5496, 1024) dtype=torch.float16 max|d|=0.000e+00 + slat_feats_noise WITHIN-FLOOR max|d|=3.383e-03 (tol 5.000e-02) + slat_feats_forced WITHIN-FLOOR max|d|=4.269e-03 (tol 5.000e-02) + [wrapped/zero-bias] + cond BIT-IDENTICAL shape=(1, 5496, 1024) dtype=torch.float16 max|d|=0.000e+00 + slat_feats_noise WITHIN-FLOOR max|d|=3.383e-03 (tol 5.000e-02) + wrapped==raw (noise) BIT-IDENTICAL shape=(4619, 8) dtype=torch.float32 max|d|=0.000e+00 + slat_feats_forced WITHIN-FLOOR max|d|=4.269e-03 (tol 5.000e-02) + wrapped==raw (forced) BIT-IDENTICAL shape=(4619, 8) dtype=torch.float32 max|d|=0.000e+00 +[ours] chair_026 (N=5858, ref steps=25 strength=1.0 rescale_t=1.0 | deployed self-noise floor=0.000e+00 tol=5.000e-02) + [raw] + cond BIT-IDENTICAL shape=(1, 5496, 1024) dtype=torch.float16 max|d|=0.000e+00 + slat_feats_noise WITHIN-FLOOR max|d|=5.830e-03 (tol 5.000e-02) + slat_feats_forced WITHIN-FLOOR max|d|=1.043e-02 (tol 5.000e-02) + [wrapped/zero-bias] + cond BIT-IDENTICAL shape=(1, 5496, 1024) dtype=torch.float16 max|d|=0.000e+00 + slat_feats_noise WITHIN-FLOOR max|d|=5.830e-03 (tol 5.000e-02) + wrapped==raw (noise) BIT-IDENTICAL shape=(5858, 8) dtype=torch.float32 max|d|=0.000e+00 + slat_feats_forced WITHIN-FLOOR max|d|=1.043e-02 (tol 5.000e-02) + wrapped==raw (forced) BIT-IDENTICAL shape=(5858, 8) dtype=torch.float32 max|d|=0.000e+00 +[ours] PARITY PASS diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/probe2_ref.pt b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/probe2_ref.pt new file mode 100644 index 0000000000000000000000000000000000000000..d2b61b86ae0e9782ad00eb8dd8f4914286b3c148 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/probe2_ref.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a9518028c75f9bb983889d290b85f83e15fc8435e2072127aab57e4c7d1a4f53 +size 10685283 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/probe_ref.pt b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/probe_ref.pt new file mode 100644 index 0000000000000000000000000000000000000000..f72f749f92392dcf8504d4ca2492fe9cb5489a74 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/probe_ref.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:033cd057451f09ac3883647d92a9685d10ca7c0d81f87dc9f1ce94724caf402d +size 43640209 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/ref.log b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/ref.log new file mode 100644 index 0000000000000000000000000000000000000000..8833c078c2898e6d8c4b44fa4b0edf0fef5bf6c8 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen/ref.log @@ -0,0 +1,73 @@ +/lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/lib/python3.11/site-packages/torch/cuda/__init__.py:61: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you. + import pynvml # type: ignore[import] +2026-08-31 18:47:31.995 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.sparse:__from_env:39 - [SPARSE] Backend: spconv, Attention: sdpa +2026-08-31 18:47:38.280 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.attention:__from_env:30 - [ATTENTION] Using backend: sdpa +[SPARSE][CONV] spconv algo: native +Warp 1.12.1 initialized: + CUDA Toolkit 12.9, Driver 13.0 + Devices: + "cpu" : "x86_64" + "cuda:0" : "NVIDIA A100-SXM4-80GB" (79 GiB, sm_80, mempool enabled) + Kernel cache: + /home/nvidia/.cache/warp/1.12.1 +2026-08-31 18:47:49.271 | INFO | sam3d_objects.pipeline.inference_pipeline:set_attention_backend:15 - GPU name is NVIDIA A100-SXM4-80GB +2026-08-31 18:47:53.396 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +2026-08-31 18:47:53.397 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +[ref] building deployed pipeline ... +2026-08-31 18:47:53.453 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +/lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/lib/python3.11/site-packages/moge/model/v1.py:172: UserWarning: The following deprecated/invalid arguments are ignored: {'output_mask': True, 'split_head': True} + warnings.warn(f"The following deprecated/invalid arguments are ignored: {deprecated_kwargs}") +2026-08-31 18:48:02.212 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +2026-08-31 18:48:02.214 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:98 - self.device: cuda +2026-08-31 18:48:02.214 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:99 - CUDA_VISIBLE_DEVICES: 1 +2026-08-31 18:48:02.214 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:100 - Actually using GPU: 0 +2026-08-31 18:48:02.214 | INFO | sam3d_objects.pipeline.inference_pipeline:init_pose_decoder:295 - Using pose decoder: ScaleShiftInvariant +2026-08-31 18:48:02.214 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:131 - Loading model weights... +2026-08-31 18:48:02.515 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/ss_generator.ckpt +2026-08-31 18:48:13.901 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_generator.ckpt +2026-08-31 18:48:19.400 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/ss_decoder.ckpt +2026-08-31 18:48:19.979 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_decoder_gs.ckpt +2026-08-31 18:48:20.456 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_decoder_gs_4.ckpt +2026-08-31 18:48:21.159 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_decoder_mesh.ckpt +2026-08-31 18:48:21.987 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:31 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-08-31 18:48:23.321 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:44 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-08-31 18:48:23.330 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:31 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-08-31 18:48:24.723 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:44 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-08-31 18:48:24.776 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/ss_generator.ckpt +2026-08-31 18:48:31.503 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:31 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-08-31 18:48:32.940 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:44 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-08-31 18:48:32.949 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:31 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-08-31 18:48:34.490 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:44 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-08-31 18:48:34.519 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_generator.ckpt +2026-08-31 18:48:40.669 | INFO | sam3d_objects.pipeline.inference_pipeline:override_ss_generator_cfg_config:520 - ss_generator parameters: inference_steps=25, cfg_strength=7, cfg_interval=[0, 500], rescale_t=3, cfg_strength_pm=0.0 +2026-08-31 18:48:40.669 | INFO | sam3d_objects.pipeline.inference_pipeline:override_slat_generator_cfg_config:542 - slat_generator parameters: inference_steps=25, cfg_strength=1, cfg_interval=[0, 500], rescale_t=1 +2026-08-31 18:48:40.669 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:196 - Loading model weights completed! +[ref] pipeline ready: slat steps=25 strength=1 interval=[0, 500] rescale_t=1 +2026-08-31 18:48:42.723 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 18:48:43.063 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 18:48:43.065 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 18:48:43.065 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 18:48:43.246 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 18:48:51.925 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 18:48:51.925 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 18:48:52.142 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +[ref] chair_026: N=5858 cond=(1, 5496, 1024)/torch.float16 feats=(5858, 8)/torch.float32 -> /lp-dev/jonghoon/mv-sam3d-6d-code/migrator/code/mv-sam3d-for-6d-v2/migrator/cache/parity_slatgen/chair_026.pt +2026-08-31 18:49:00.060 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 18:49:00.226 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 18:49:00.227 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 18:49:00.227 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 18:49:00.389 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 18:49:09.140 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 18:49:09.140 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 18:49:09.316 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +[ref] bunny_003: N=7984 cond=(1, 5496, 1024)/torch.float16 feats=(7984, 8)/torch.float32 -> /lp-dev/jonghoon/mv-sam3d-6d-code/migrator/code/mv-sam3d-for-6d-v2/migrator/cache/parity_slatgen/bunny_003.pt +2026-08-31 18:49:18.333 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 18:49:18.496 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 18:49:18.497 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 18:49:18.497 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 18:49:18.660 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 18:49:26.196 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 18:49:26.196 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 18:49:26.401 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +[ref] bunny_008: N=4619 cond=(1, 5496, 1024)/torch.float16 feats=(4619, 8)/torch.float32 -> /lp-dev/jonghoon/mv-sam3d-6d-code/migrator/code/mv-sam3d-for-6d-v2/migrator/cache/parity_slatgen/bunny_008.pt +[ref] DONE diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/bunny_003.pt b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/bunny_003.pt new file mode 100644 index 0000000000000000000000000000000000000000..fe5d5d168ce6438a6e9cc22b11b365979babe464 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/bunny_003.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:14502f014a45dee714dbd919e394b419a72e95419897303bc7386eed8e0a5508 +size 20996178 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/bunny_008.pt b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/bunny_008.pt new file mode 100644 index 0000000000000000000000000000000000000000..660af17266cfbe0dd826aac7ce2778b44490a98e --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/bunny_008.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92e4ff59ac962b00dc5a5975913c5e952ece70ab3a640d25c0727bd68265ae08 +size 20511442 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/chair_026.pt b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/chair_026.pt new file mode 100644 index 0000000000000000000000000000000000000000..bf508f1b8bac54743652d0a3b90623da9a8c7920 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/chair_026.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bdf5dc60ce5b61d76bb78762bd5895cdb21ff3a49252f63d9ef8007c73ed3ffd +size 20690002 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/ref.log b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/ref.log new file mode 100644 index 0000000000000000000000000000000000000000..233f65e5789113ddfa7383a2864c12d57d72713a --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/parity_slatgen_b/ref.log @@ -0,0 +1,73 @@ +/lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/lib/python3.11/site-packages/torch/cuda/__init__.py:61: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you. + import pynvml # type: ignore[import] +2026-08-31 19:04:42.766 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.sparse:__from_env:39 - [SPARSE] Backend: spconv, Attention: sdpa +2026-08-31 19:04:47.903 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.attention:__from_env:30 - [ATTENTION] Using backend: sdpa +[SPARSE][CONV] spconv algo: native +Warp 1.12.1 initialized: + CUDA Toolkit 12.9, Driver 13.0 + Devices: + "cpu" : "x86_64" + "cuda:0" : "NVIDIA A100-SXM4-80GB" (79 GiB, sm_80, mempool enabled) + Kernel cache: + /home/nvidia/.cache/warp/1.12.1 +2026-08-31 19:04:56.352 | INFO | sam3d_objects.pipeline.inference_pipeline:set_attention_backend:15 - GPU name is NVIDIA A100-SXM4-80GB +2026-08-31 19:04:58.555 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +2026-08-31 19:04:58.555 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +[ref] building deployed pipeline ... +2026-08-31 19:04:58.594 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +/lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/lib/python3.11/site-packages/moge/model/v1.py:172: UserWarning: The following deprecated/invalid arguments are ignored: {'output_mask': True, 'split_head': True} + warnings.warn(f"The following deprecated/invalid arguments are ignored: {deprecated_kwargs}") +2026-08-31 19:05:08.207 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +2026-08-31 19:05:08.209 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:98 - self.device: cuda +2026-08-31 19:05:08.209 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:99 - CUDA_VISIBLE_DEVICES: 1 +2026-08-31 19:05:08.209 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:100 - Actually using GPU: 0 +2026-08-31 19:05:08.209 | INFO | sam3d_objects.pipeline.inference_pipeline:init_pose_decoder:295 - Using pose decoder: ScaleShiftInvariant +2026-08-31 19:05:08.209 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:131 - Loading model weights... +2026-08-31 19:05:08.653 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/ss_generator.ckpt +2026-08-31 19:05:20.732 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_generator.ckpt +2026-08-31 19:05:25.889 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/ss_decoder.ckpt +2026-08-31 19:05:26.449 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_decoder_gs.ckpt +2026-08-31 19:05:26.881 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_decoder_gs_4.ckpt +2026-08-31 19:05:27.467 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_decoder_mesh.ckpt +2026-08-31 19:05:28.226 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:31 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-08-31 19:05:30.056 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:44 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-08-31 19:05:30.065 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:31 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-08-31 19:05:31.477 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:44 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-08-31 19:05:31.522 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/ss_generator.ckpt +2026-08-31 19:05:39.237 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:31 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-08-31 19:05:40.567 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:44 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-08-31 19:05:40.576 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:31 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-08-31 19:05:41.857 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:44 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-08-31 19:05:41.882 | INFO | sam3d_objects.model.io:load_model_from_checkpoint:158 - Loading checkpoint from checkpoints/hf/slat_generator.ckpt +2026-08-31 19:05:48.113 | INFO | sam3d_objects.pipeline.inference_pipeline:override_ss_generator_cfg_config:520 - ss_generator parameters: inference_steps=25, cfg_strength=7, cfg_interval=[0, 500], rescale_t=3, cfg_strength_pm=0.0 +2026-08-31 19:05:48.114 | INFO | sam3d_objects.pipeline.inference_pipeline:override_slat_generator_cfg_config:542 - slat_generator parameters: inference_steps=25, cfg_strength=1, cfg_interval=[0, 500], rescale_t=1 +2026-08-31 19:05:48.114 | INFO | sam3d_objects.pipeline.inference_pipeline:__init__:196 - Loading model weights completed! +[ref] pipeline ready: slat steps=25 strength=1 interval=[0, 500] rescale_t=1 +2026-08-31 19:05:48.971 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 19:05:49.265 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 19:05:49.267 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 19:05:49.267 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 19:05:49.431 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 19:05:57.635 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 19:05:57.635 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 19:05:57.803 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +[ref] chair_026: N=5858 cond=(1, 5496, 1024)/torch.float16 feats=(5858, 8)/torch.float32 -> /lp-dev/jonghoon/mv-sam3d-6d-code/migrator/code/mv-sam3d-for-6d-v2/migrator/cache/parity_slatgen_b/chair_026.pt +2026-08-31 19:06:05.500 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 19:06:05.675 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 19:06:05.676 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 19:06:05.676 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 19:06:05.827 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 19:06:14.623 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 19:06:14.623 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 19:06:14.843 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +[ref] bunny_003: N=7984 cond=(1, 5496, 1024)/torch.float16 feats=(7984, 8)/torch.float32 -> /lp-dev/jonghoon/mv-sam3d-6d-code/migrator/code/mv-sam3d-for-6d-v2/migrator/cache/parity_slatgen_b/bunny_003.pt +2026-08-31 19:06:22.977 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 19:06:23.201 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 19:06:23.202 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 19:06:23.202 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 19:06:23.378 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +2026-08-31 19:06:30.672 | INFO | sam3d_objects.pipeline.inference_pipeline:sample_slat:860 - Sampling sparse latent: inference_steps=25, strength=1, interval=[0, 500], rescale_t=1 +2026-08-31 19:06:30.672 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:748 - Running condition embedder ... +2026-08-31 19:06:30.837 | INFO | sam3d_objects.pipeline.inference_pipeline:get_condition_input:752 - Condition embedder finishes! +[ref] bunny_008: N=4619 cond=(1, 5496, 1024)/torch.float16 feats=(4619, 8)/torch.float32 -> /lp-dev/jonghoon/mv-sam3d-6d-code/migrator/code/mv-sam3d-for-6d-v2/migrator/cache/parity_slatgen_b/bunny_008.pt +[ref] DONE diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/smoke_slatflow.log b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/smoke_slatflow.log new file mode 100644 index 0000000000000000000000000000000000000000..c7983f424b2175734a7862aa72c00b34cc05bc2e --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/smoke_slatflow.log @@ -0,0 +1,19 @@ +/lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/lib/python3.11/site-packages/torch/cuda/__init__.py:61: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you. + import pynvml # type: ignore[import] +2026-09-01 01:55:05.629 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.sparse:__from_env:39 - [SPARSE] Backend: spconv, Attention: sdpa +2026-09-01 01:55:09.659 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.attention:__from_env:30 - [ATTENTION] Using backend: sdpa +2026-09-01 01:55:15.109 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +[SPARSE][CONV] spconv algo: native +[smoke] object=chair_026 N=5858 n_subset=2 views=[0, 1, 2, 3] crops=(4, 3, 518, 518) uv=(4, 5858, 2) vis=(4, 5858) plucker=(1, 1369, 6) bias_views=1 +[slat-pretrained] slat_generator: loaded 526/526 own tensors from '_base_models.generator.*' (0 missing, 0 unexpected) +2026-09-01 01:55:25.930 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:32 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-09-01 01:55:31.714 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:51 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-09-01 01:55:31.718 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:32 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-09-01 01:55:36.823 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:51 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +[slat-pretrained] slat_condition_embedder: loaded 699/699 own tensors from '_base_models.condition_embedder.*' (0 missing, 0 unexpected) +[smoke] model built: trainable=617.7M bias_enabled=True +[smoke] flow_step: loss=0.7099 frac_visible_x0=0.510 frac_target=0.702 +[smoke] grad norms: backbone=6.378e+01 fuser=2.272e+00 plucker_fc=3.835e-02 +[smoke] bias: shape=(1394, 6865) anchor_max|.|=0.0 nonanchor_vals=[-10.0, 0.0] frac_open=0.0066 +[smoke] sample: cond=(1, 6865, 1024) base=(1, 5858, 8) slat feats=(5858, 8) finite=True std=4.557 +[smoke] SELF-CHECK PASS diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/train_slatflow_smoke.log b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/train_slatflow_smoke.log new file mode 100644 index 0000000000000000000000000000000000000000..e6924e6b4f462cdb3a27da078f74b9606238c009 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/migrator/cache/train_slatflow_smoke.log @@ -0,0 +1,28 @@ +/lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/lib/python3.11/site-packages/torch/cuda/__init__.py:61: FutureWarning: The pynvml package is deprecated. Please install nvidia-ml-py instead. If you did not install pynvml directly, please report this to the maintainers of the package that installed pynvml for you. + import pynvml # type: ignore[import] +2026-09-01 01:56:18.594 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.sparse:__from_env:39 - [SPARSE] Backend: spconv, Attention: sdpa +2026-09-01 01:56:22.431 | INFO | sam3d_objects.model.backbone.tdfy_dit.modules.attention:__from_env:30 - [ATTENTION] Using backend: sdpa +[SPARSE][CONV] spconv algo: native +[train] train samples: 201 +[train] val samples: 100 +[slat-pretrained] slat_generator: loaded 526/526 own tensors from '_base_models.generator.*' (0 missing, 0 unexpected) +2026-09-01 01:56:35.604 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:32 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-09-01 01:56:40.418 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:51 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +2026-09-01 01:56:40.422 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:32 - Loading DINO model: dinov2_vitl14_reg from facebookresearch/dinov2 (source: github) +2026-09-01 01:56:45.101 | INFO | sam3d_objects.model.backbone.dit.embedder.dino:__init__:51 - Loaded DINO model - type: , embed_dim: 1024, patch_size: (14, 14) +[slat-pretrained] slat_condition_embedder: loaded 699/699 own tensors from '_base_models.condition_embedder.*' (0 missing, 0 unexpected) +[setup] trainable params: 617.7M / 1226.5M (enable_bias=True, fp16_torso=False) +/lp-dev/jonghoon/mv-sam3d-6d-code/migrator/code/mv-sam3d-for-6d-v2/mvsam3d/train/train_slat_flow.py:277: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead. + scaler = torch.cuda.amp.GradScaler(enabled=use_scaler) +[setup] backend=ddp world=1 global_batch=2 (objects/optim-step) steps=2 grad_accum=2 amp=True/bf16 val_every=2 +2026-09-01 01:56:54.846 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +2026-09-01 01:56:54.854 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +2026-09-01 01:56:54.893 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +2026-09-01 01:56:54.945 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +[train] step 1 lr=4.000000e-07 frac_target=0.7416 frac_visible_x0=0.6616 loss=0.9653 loss/slat=0.9653 (0.14 it/s) +[train] step 2 lr=6.000000e-07 frac_target=0.7212 frac_visible_x0=0.4989 loss=1.9295 loss/slat=1.9295 (0.48 it/s) +2026-09-01 01:57:04.080 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +2026-09-01 01:57:04.091 | WARNING | sam3d_objects.data.dataset.tdfy.preprocessor:__post_init__:51 - No rgb pointmap normalizer provided, using scale + shift +[val] step 2 frac_target=0.5975 frac_visible_x0=0.4419 loss=0.9582 loss/slat=0.9582 (n=2) +[train] saved /lp-dev/jonghoon/mv-sam3d-6d-code/migrator/cache/ckpt_slatflow/step_0000002.pt +[train] done. diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/__init__.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/__init__.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..edc544bbb4a3131d2d6a5a2bf0315d0bbe8030f0 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/__init__.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..06301a0dc82171b9a8f14ea9e7a94a7cf70de45e Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/__init__.cpython-39.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/__init__.cpython-39.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7871904f89a5261b88e00974ead8e79e41166793 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/__init__.cpython-39.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/paths.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/paths.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b5f1a6a7134f418b97ba15cc042530cc23643ed4 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/paths.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/paths.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/paths.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..cda1e05f238fd698b365d12e4ab7a673e30112a5 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/paths.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/paths.cpython-39.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/paths.cpython-39.pyc new file mode 100644 index 0000000000000000000000000000000000000000..afd303096769339491c42b34eb6f54dd9abd3647 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/__pycache__/paths.cpython-39.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/__init__.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/__init__.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1c556b2d8f5b9cc9a80047da16f6d1c5c2b9f140 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/__init__.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..81f0c076a3fe3d61fbb73ef9b8e057c61ab3cb74 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/__init__.cpython-39.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/__init__.cpython-39.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1f17d48a4b89dff7c5a3b70fc5419736f7e57bbc Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/__init__.cpython-39.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/adapters.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/adapters.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..af05cbc936f09e41cac369aeba92fccb85870e37 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/adapters.cpython-311.pyc @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:859d14326540e24088b672bc15940079ab435832247d5921114087cdbcb3a342 +size 118636 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/base_dataset.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/base_dataset.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3f4dc4c0056f3ff3a28518d9b06d06375048fed4 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/base_dataset.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/canonical.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/canonical.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b7f4c3f50478f85a223d6f75e0272643b98de93a Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/canonical.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/dexycb_calib.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/dexycb_calib.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..95c0902a2a40292cf24a878091018b2bef7bd2d5 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/dexycb_calib.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/dexycb_dataset.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/dexycb_dataset.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..97d0a4f7d3d67ff2027ff532deb5fe34f457f749 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/dexycb_dataset.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/integrated.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/integrated.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9746874dcc89c7ee6b97fadbe9549764f352d000 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/integrated.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slat_preprocess.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slat_preprocess.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..51f304e97b48347d61a15cac466b52646539e858 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slat_preprocess.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slatflow_dataset.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slatflow_dataset.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9e80f5c43b3c80d80bc51564156de26b5f25e48c Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slatflow_dataset.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slatflow_dataset.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slatflow_dataset.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a323472da5e21680acf58173098660a7d6ddef4c --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/slatflow_dataset.cpython-311.pyc @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:920741be4f6ca85689c3629a6d65f832b0a060dbc701db71a0a7bcfd131e179e +size 126300 diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ss_imgcond_dataset.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ss_imgcond_dataset.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1a48974cd39c1220497946cb1279fa19befa7665 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ss_imgcond_dataset.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ss_imgcond_dataset.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ss_imgcond_dataset.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a9c44d78cfc131d773cdf86d0789cedfbeacc616 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ss_imgcond_dataset.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ss_preprocess.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ss_preprocess.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b699767377cecb48e63e07823fb0caf227c9c9c4 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ss_preprocess.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ssflow_dataset.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ssflow_dataset.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b2a74bbe33df9596f60de6d8316a7f7769f86e36 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ssflow_dataset.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ssflow_dataset.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ssflow_dataset.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..5c7be673d551b0b24039fdb25a9e2331b46adebd Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ssflow_dataset.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ssflow_dataset.cpython-39.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ssflow_dataset.cpython-39.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9ec58a496becab77b1403d52c63f98b3ee01ded6 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/ssflow_dataset.cpython-39.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/v_bucket_sampler.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/v_bucket_sampler.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b7c68bb96fb0d03e5f436856dad55e5513b9894e Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/__pycache__/v_bucket_sampler.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/slatflow_dataset.py b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/slatflow_dataset.py index 11262a8737b96f97194fbe748509f361ef4af93e..9a5517d8c07dc49af2002b467446f7ce77a77364 100644 --- a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/slatflow_dataset.py +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/slatflow_dataset.py @@ -903,6 +903,53 @@ class TrellisSlatFlowDataset(TrellisSSFlowDataset): } +# ---- RecGen camera QC (2026-09-24) ------------------------------------------ # +# A RecGen view's camera is VALID iff the object's OWN SLAT voxels, projected with +# that view's camera, land inside RecGen's (modal) mask in proportion to how much +# of the object RecGen says is visible: +# p = #voxels projecting inside the mask / #voxels (all voxels) +# vf = RecGen view_metadata visible_fraction (modal / amodal) +# valid <=> vf >= RG_VF_MIN and p >= RG_RATIO * vf +# With a correct pose p ~= vf (occluders remove voxels from the modal mask exactly +# as they remove visible area). Calibrated on 4,000 random RecGen sets against our +# latents: keeps 99.0% of views of trusted analytic poses (p/vf p01 = 0.72), 0% of +# the analytic poses that failed their own seed test. Object guard: if fewer than +# RG_OBJ_MIN_FRAC of the object's verifiable (vf >= RG_VF_MIN) views pass, the pose +# is treated as systematically wrong and ALL its RecGen views are dropped (the +# object falls back to white-only, or is skipped if it has no white cond). +RG_VF_MIN = 0.10 +RG_RATIO = 0.60 +RG_OBJ_MIN_FRAC = 0.50 + + +def recgen_view_qc(coords3: np.ndarray, views_all: List[Dict], masks: List[np.ndarray], + vfs: Sequence[float], vf_min: float = RG_VF_MIN, + ratio: float = RG_RATIO, obj_min_frac: float = RG_OBJ_MIN_FRAC): + """-> (valid view indices, per-view p). coords3 (N,3) int voxel coords of the + object's SLAT latent; views_all[i] has fx,fy,cx,cy,c2w (rigid, SLAT cube); + masks[i] (H,W) bool; vfs[i] RecGen visible_fraction.""" + pts = (np.asarray(coords3, dtype=np.float64) + 0.5) / VOX - 0.5 + N = max(len(pts), 1) + ps, ok = [], [] + for v, m, vf in zip(views_all, masks, vfs): + w2c = np.linalg.inv(v["c2w"]) + xc = pts @ w2c[:3, :3].T + w2c[:3, 3] + z = xc[:, 2] + zs = np.where(z == 0, 1e-12, z) + u = np.round(v["fx"] * xc[:, 0] / zs + v["cx"]).astype(np.int64) + q = np.round(v["fy"] * xc[:, 1] / zs + v["cy"]).astype(np.int64) + H, W = m.shape[:2] + inf = (z > 0) & (u >= 0) & (u < W) & (q >= 0) & (q < H) + p = float(m[q[inf], u[inf]].sum()) / N + ps.append(p) + ok.append(bool(vf >= vf_min and p >= ratio * vf)) + verifiable = sum(1 for vf in vfs if vf >= vf_min) + valid = [i for i, o in enumerate(ok) if o] + if not valid or len(valid) < obj_min_frac * verifiable: + valid = [] + return valid, ps + + DEFAULT_TAR_ROOTS = ("/data/mv_mesh_data/slat_tars/processed",) DEFAULT_TAR_MANIFESTS = "/data/mv_mesh_data/DATASET_INDEX/hubman/processed/manifests" @@ -960,30 +1007,33 @@ class TarSlatFlowDataset(TrellisSlatFlowDataset): shards: Optional[Sequence[str]] = None, max_objects: Optional[int] = None, use_recgen_images: bool = False, - p_recgen_image: float = 0.0, cond_cameras_fallback: bool = False): - # ---- condition-variant knobs (2026-09-24; all OFF by default -> the - # index, the draws and every item are byte-identical to the legacy path) + # ---- condition-source knobs (2026-09-24). Class defaults OFF -> index, + # draws and every item byte-identical to the legacy loader. The PROD + # config (train_slatflow_prod.yaml) turns BOTH ON. # use_recgen_images : also index ``recgen//{cameras.json,NN.jpg, # NN_mask.png}`` members (RecGen's own background-baked RGB + its # visible mask + per-view RIGID OpenCV c2w in the SLAT cube) from ANY - # tar under tar_roots, joined to ``latents/.npz`` by id. Objects - # with a latent + RecGen images but NO white cond (latent-only new - # slats) become trainable instead of being silently skipped. - # p_recgen_image : when an object has BOTH sources, P(use RecGen) per - # (sample, epoch) on its own rng stream (offset 9191). Objects with - # only one source always use it. - # RecGen variant = the SAME default path with the RecGen full-frame - # image (background kept, alpha = RecGen mask) substituted for the - # white cond render in the preprocess_slat_image slot (item["image"]); + # tar under tar_roots, joined to ``latents/.npz`` by the SAME id + # string. Conditioning then follows the ONE-POOL rule: + # * an object with BOTH sources draws its anchor uniformly from the + # union of its (valid) white cond views and (valid) RecGen views; + # * the anchor's source decides everything: ALL auxiliary views of + # the sample come from that same source (and its cameras) — a + # sample never mixes white and RecGen views; + # * single-source objects draw exactly as the legacy loader does on + # the source they have (white-only items are byte-identical to + # legacy; latent-only RecGen slats become trainable). + # A RecGen anchor goes through the SAME default path: the full-frame + # RecGen image (background kept, alpha = its mask) is fed to + # preprocess_slat_image -> item["image"] (+ mask/rgb_image/...); # DINO seed crops / uv / vis use the mask-matted tight crop exactly - # like the white path, cameras = RecGen per-view c2w. + # like the white path. # cond_cameras_fallback : white-cond objects lacking # ``cameras//cond_cameras.json`` (the 24-view slat5/slat5rs # batches) use ``cond//transforms.json`` (same transform_matrix; # fx=fy=W/(2 tan(camera_angle_x/2)), cx=cy=W/2 from the PNG width). self.use_recgen_images = bool(use_recgen_images) - self.p_recgen_image = float(p_recgen_image) self.cond_cameras_fallback = bool(cond_cameras_fallback) # NOTE: deliberately does NOT call TrellisSlatFlowDataset.__init__ (that # scans object DIRS). We set the attributes its shared methods @@ -1062,7 +1112,7 @@ class TarSlatFlowDataset(TrellisSlatFlowDataset): # new index semantics -> distinct cache signature (legacy sig # unchanged when both knobs are off, so old caches stay valid) parts.append(f"rg={int(self.use_recgen_images)}" - f":fb={int(self.cond_cameras_fallback)}:v1") + f":fb={int(self.cond_cameras_fallback)}:v2") return _hashlib.sha1("|".join(parts).encode()).hexdigest() _sig = _tar_sig(tar_files) _cache = os.environ.get("TAR_INDEX_CACHE") or os.path.join( @@ -1290,9 +1340,27 @@ class TarSlatFlowDataset(TrellisSlatFlowDataset): views_all.append(dict(depth=None, fx=float(K[0, 0]), fy=float(K[1, 1]), cx=float(K[0, 2]), cy=float(K[1, 2]), c2w=np.asarray(c["c2w"], dtype=np.float64), - res=int(max(int(c["width"]), int(c["height"]))))) + res=int(max(int(c["width"]), int(c["height"]))), + vf=float(c.get("visible_fraction", -1.0)))) return views_all, len(views_all) + def _rg_qc_valid(self, sha: str, m: Dict, views_all: List[Dict], + sc: np.ndarray) -> List[int]: + """RecGen views passing the camera QC (recgen_view_qc), cached per object.""" + key = (sha, "rgqc") + v = self._valid_cache.get(key) + if v is not None: + return v + r = m["rg"] + tar = self._get_tar(r["tar"]) + masks = [np.array(Image.open(io.BytesIO(tar.extractfile(n).read())).convert("L")) > 127 + for n in r["masks"]] + valid, _ = recgen_view_qc(sc, views_all, masks, [vw["vf"] for vw in views_all]) + if len(self._valid_cache) > 20000: + self._valid_cache.clear() + self._valid_cache[key] = valid + return valid + def _load_view_rgba(self, m: Dict, src: str, i: int) -> np.ndarray: """FULL-FRAME RGBA uint8 of view i. white: the cond render PNG. recgen: RecGen's RGB (background kept) + its visible mask as alpha (>127).""" @@ -1307,18 +1375,6 @@ class TarSlatFlowDataset(TrellisSlatFlowDataset): return np.concatenate([rgb, a], axis=2) return self._load_png_rgba(self._get_tar(m["tar"]), m["cond_pngs"][i]) - def _cond_source(self, idx: int, m: Dict) -> str: - """'white' | 'recgen' for this (sample, epoch). Own rng stream (9191).""" - has_w = bool(m.get("cond_pngs")) - has_r = m.get("rg") is not None - if has_r and not has_w: - return "recgen" - if has_r and has_w and self.p_recgen_image > 0.0: - r = float(np.random.default_rng( - [self.seed, int(self._epoch), int(idx), 9191]).random()) - return "recgen" if r < self.p_recgen_image else "white" - return "white" - # ---- OPT-IN bad-view gate (tar variant, keyed by sha) -------------- # def _tar_valid_views(self, sha: str, tar, views_all, sc, centers, V, src: str = "white") -> List[int]: @@ -1381,11 +1437,17 @@ class TarSlatFlowDataset(TrellisSlatFlowDataset): tar = self._get_tar(m["tar"]) rng = np.random.default_rng([self.seed, self._epoch, idx]) - src = self._cond_source(idx, m) # 'white' unless recgen knobs on - if src == "recgen": - views_all, V = self._rg_cameras(m) - else: - views_all, V = self._tar_cameras(tar, m) + # ---- condition sources present for this object ------------------- # + # white = cond//NNN.png + its cameras (always, legacy objects) + # recgen = recgen//NN.jpg + NN_mask.png + cameras.json + # (only indexed when use_recgen_images=True) + has_w = bool(m.get("cond_pngs")) + has_r = m.get("rg") is not None + cams = {} + if has_w: + cams["white"] = self._tar_cameras(tar, m) + if has_r: + cams["recgen"] = self._rg_cameras(m) # x1 = released SLAT: coords 0..63 used DIRECTLY, feats row-aligned. zl = np.load(io.BytesIO(tar.extractfile(m["latent"]).read())) @@ -1398,15 +1460,64 @@ class TarSlatFlowDataset(TrellisSlatFlowDataset): torch.from_numpy(sc).int()], dim=1) # (N,4) int32 centers = (sc.astype(np.float64) + 0.5) / VOX - 0.5 - n_sub, no_seed, mode = self._mixture_n_sub(idx, V, rng) - if self.filter_bad_views: # OPT-IN: draw only from valid - valid_arr = np.asarray( - self._tar_valid_views(sha, tar, views_all, sc, centers, V, src), - dtype=int) - n_sub = min(int(n_sub), len(valid_arr)) - sub = rng.choice(valid_arr, size=n_sub, replace=False) # anchor in valid + # RecGen camera QC (always on for RecGen views): only views whose camera + # projects this object's own voxels into RecGen's mask consistently with + # RecGen's visible_fraction may enter the pool. None valid -> the + # object's RecGen set is dropped (white-only fallback, or skipped). + rg_ok = None + if has_r: + rg_ok = self._rg_qc_valid(sha, m, cams["recgen"][0], sc) + if not rg_ok: + has_r = False + if not (has_w or has_r): + raise ValueError(f"{sha}: no condition view passes (white cond absent, " + f"all RecGen views fail camera QC)") + + def _valid_for(s): + va, Vs = cams[s] + fb = (np.asarray(self._tar_valid_views(sha, tar, va, sc, centers, Vs, s), + dtype=int) if self.filter_bad_views + else np.arange(Vs, dtype=int)) + if s == "recgen": # QC is the hard rule + both = np.asarray([i for i in fb if i in set(rg_ok)], dtype=int) + fb = both if len(both) else np.asarray(rg_ok, dtype=int) + return fb + + if not (has_w and has_r): + # SINGLE source (white-only = every legacy object; recgen-only = the + # latent-only new slats): the exact legacy draw on that source. + src = "white" if has_w else "recgen" + views_all, V = cams[src] + n_sub, no_seed, mode = self._mixture_n_sub(idx, V, rng) + if src == "recgen": # QC-valid (+ bad-view gate) only + valid_arr = _valid_for("recgen") + n_sub = min(int(n_sub), len(valid_arr)) + sub = rng.choice(valid_arr, size=n_sub, replace=False) + elif self.filter_bad_views: # OPT-IN: draw only from valid + valid_arr = np.asarray( + self._tar_valid_views(sha, tar, views_all, sc, centers, V, src), + dtype=int) + n_sub = min(int(n_sub), len(valid_arr)) + sub = rng.choice(valid_arr, size=n_sub, replace=False) # anchor in valid + else: + sub = rng.choice(V, size=n_sub, replace=False) else: - sub = rng.choice(V, size=n_sub, replace=False) + # BOTH sources -> ONE POOL. Candidate anchors = the union of the + # object's (valid) white cond views and (QC-valid) RecGen views; the + # anchor is drawn uniformly from the pool, and the ANCHOR'S SOURCE + # DECIDES EVERYTHING: all auxiliary views are drawn from that same + # source only (with that source's cameras) — never mixed. + valid = {s: _valid_for(s) for s in ("white", "recgen")} + pool = ([("white", int(i)) for i in valid["white"]] + + [("recgen", int(i)) for i in valid["recgen"]]) + n_sub, no_seed, mode = self._mixture_n_sub(idx, len(pool), rng) + src, a_i = pool[int(rng.integers(len(pool)))] + views_all, V = cams[src] + rest = np.asarray([i for i in valid[src] if i != a_i], dtype=int) + n_sub = min(int(n_sub), 1 + len(rest)) + aux = (rng.choice(rest, size=n_sub - 1, replace=False) if n_sub > 1 + else np.zeros(0, dtype=int)) + sub = np.concatenate([[a_i], aux]).astype(int) order = list(sub) + [i for i in range(V) if i not in set(sub.tolist())] crops, uvs, viss, fulls = [], [], [], [] diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/slatflow_dataset.py.bak_predatveiw b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/slatflow_dataset.py.bak_predatveiw new file mode 100644 index 0000000000000000000000000000000000000000..986a9f5d2c0901a02552a597136f530154792f93 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/data/slatflow_dataset.py.bak_predatveiw @@ -0,0 +1,804 @@ +"""TrellisSlatFlowDataset — per-view DINO inputs + cameras + SLAT target on top +of TrellisSSFlowDataset (SLAT_GEN_PLAN §8). + +Reused verbatim from the SS dataset: object scan / meta.json completion marker / +split logic, the per-(sample, epoch) view-subset RNG, and the views.npz loader +(per-view K + c2w_cv — those cameras cover BOTH the Plücker rays and the +voxel->patch projection; no new camera fields). + +Per __getitem__ (CPU only — all frozen-GPU work happens in the train step): + * coords_slat = downsample(prune(voxels.npz coords, dist=1)) — the exact + inference coordinate ops (pure coordinate transforms, run on CPU torch); + * a view subset S (|S| ~ U{min_views..max_views}); anchor = S[0]; view order + of every per-view tensor = [S..., complement...] so the model can slice + the subset (x0/cond) vs all views; + * per view: the RGBA crop in the STORED bbox frame -> 518 LANCZOS -> + premult-alpha-on-black float [3,518,518] (dino_features convention, + batch_appforce:253-264) — feeds the shared frozen DINO forward; + * per view: (uv, vis) — voxel-center projections through the ABSOLUTE camera + (project_visible: nearest-pixel depth test, TOL=0.02) with uv already in + the crop-frame [-1,1] grid_sample convention (sample_feats :267-275), + computed in float64 and cast to float32; + * anchor: the 4 slat condition tensors via the DEPLOYED slat_preprocessor + recipe (OQ-6) on the raw RGBA input image; + * non-anchor: per-patch anchor-relative Plücker rays [1369,6] (§4a) and the + soft-mask camera dicts (§4b); + * x1: ``slat/slat_official_sam3d.npz`` — the OFFICIAL TRELLIS-toolkit + latent with SAM3D's slat_encoder (RAW, all active voxels; produced by + ``tools/gen_x1_official.py``). Rows are coord-hash-matched onto + coords_slat (a prune-subset of the voxel set) and full coverage is + asserted. There is NO on-the-fly fallback: the former full-view + visible-mean target (reimplemented, non-official) was deleted — the + train step errors if the file is absent. + +RGB SOURCE: the production layout stores ``views/rgb/{i:03d}.webp`` + a +``views/bbox.npy`` sidecar. The toys4k SMOKE objects pre-date that layout, so +when those files are missing the loader falls back to the toys4k1k render +assets (inputs/_.png = the bbox crop, renders/_.npz for +bbox/res) with the fixed front/side/oside/back view order — the exact files +batch_appforce consumed. +""" +from __future__ import annotations + +import os +from typing import Dict, List, Optional, Sequence + +import numpy as np +import torch +from PIL import Image + +from mvsam3d.data.ssflow_dataset import TrellisSSFlowDataset, load_views_npz +from mvsam3d.data.slat_preprocess import preprocess_slat_image +from mvsam3d.model.mv_slat_condition import plucker_rays_for_view + +TOL = 0.02 # visibility depth tolerance (batch_appforce) +VOX = 64 +# SS-stage seed-mixture mode -> integer code carried in the item/batch (mirrors +# ss_imgcond_dataset._MODE_CODE; 'val' shares 'single'==1 as an eval label). +_SEED_MODE_CODE = {"none": 0, "single": 1, "multi": 2, "legacy": 3, "val": 1} +TOYS4K1K = "/lp-dev/jonghoon/mv-mesh/exp_faithfulness/toys4k1k" +TOYS4K_VIEW_ORDER = ("front", "side", "oside", "back") +# TRELLIS-style per-object voxel cap (structured_latent.py filter_metadata): +# objects with MORE active 64^3 voxels than this are DROPPED at index-build +# time (never downsampled — downsample_sparse_structure is inference-only and +# corrupts training targets). 20000 = the single-object SLAT OOM-safe bound +# (PLAN/BATCH_SIZE_BENCH.md; OOM at N~24000). Drops ~4.8% of trellis500k. +MAX_NUM_VOXELS_DEFAULT = 20000 +# visibility source when a view has no rendered depth (production data): +# raymarch = exact self-occlusion through the 64^3 grid (default) +# zbuffer = image-space min-splat (biased ~1 voxel near; see data/vis_check) +# Calibrated against rendered-depth GT on 301 toys4k objects x 4 views +# (data/vis_check/vis_check4.py, run4.log; agreement / IoU vs project_visible_np): +# zbuffer tol .03 .8762 / .6692 raymarch skip 1.5 .8113 / .4355 +# raymarch skip 2.0 .8742 / .6646 raymarch skip 2.5 .8755 / .6724 <- default +# raymarch skip 3.0 .8519 / .6472 ray2.0 OR zbuf .8757 / .6729 +# skip_vox 2.5 mirrors the GT's own TOL=0.02 slack (~1.3 voxels) plus the +# half-voxel centre offset; it matches the z-buffer's accuracy without the +# z-buffer's splat-radius/resolution heuristic. ~40 ms/view (100 ms under a +# 48-way loaded box) at N=20000. +VIS_METHOD = os.environ.get("MVSAM3D_VIS_METHOD", "raymarch") +VIS_SKIP_VOX = float(os.environ.get("MVSAM3D_VIS_SKIP_VOX", "2.5")) +VIS_ZBUF_TOL = float(os.environ.get("MVSAM3D_VIS_ZBUF_TOL", "0.03")) + + +def num_active_voxels(voxels_npz: str) -> int: + """Active 64^3 voxel count = #rows of voxels.npz 'coords'. + + Reads only the .npy header inside the zip archive (streaming decompression + of a few hundred bytes) — cheap enough to run over the whole index at + dataset-build time. Falls back to a full np.load on any header oddity.""" + import zipfile + from numpy.lib import format as npfmt + try: + with zipfile.ZipFile(voxels_npz) as zf: + with zf.open("coords.npy") as f: + version = npfmt.read_magic(f) + shape, _, _ = npfmt._read_array_header(f, version) + return int(shape[0]) + except Exception: + return int(np.load(voxels_npz)["coords"].shape[0]) + + +# --------------------------------------------------------------------------- # +def slat_coords_from_voxels(coords3: np.ndarray) -> torch.Tensor: + """voxels.npz coords (N,3) -> the SLAT coord set (M,4) int32 (batch col 0): + prune_sparse_structure(dist=1) + downsample_sparse_structure — the exact + inference ops (inference_pipeline.py:818-833), on CPU.""" + from sam3d_objects.pipeline.inference_utils import ( + downsample_sparse_structure, prune_sparse_structure) + c = torch.from_numpy(np.asarray(coords3, dtype=np.int64)).int() + c4 = torch.cat([torch.zeros(len(c), 1, dtype=torch.int32), c], dim=1) + c4 = prune_sparse_structure(c4, max_neighbor_axes_dist=1) + c4, _ = downsample_sparse_structure(c4) + return c4.int() + + +def project_visible_np(centers: np.ndarray, depth: np.ndarray, fx, fy, cx, cy, + c2w: np.ndarray, res: int): + """batch_appforce_sam3d.project_visible (:236-250), verbatim numpy port. + centers (N,3) float64 canonical -> (u, v, z, visible).""" + w2c = np.linalg.inv(c2w) + R, t = w2c[:3, :3], w2c[:3, 3] + xc = centers @ R.T + t + z = xc[:, 2] + u = fx * xc[:, 0] / z + cx + vv = fy * xc[:, 1] / z + cy + ui = np.round(u).astype(int) + vi = np.round(vv).astype(int) + inframe = (z > 0) & (ui >= 0) & (ui < res) & (vi >= 0) & (vi < res) + dep = np.zeros(len(centers)) + dep[inframe] = depth[vi[inframe], ui[inframe]] + visible = inframe & (dep > 0) & (np.abs(z - dep) < TOL) + return u, vv, z, visible + + +def zbuffer_visible(centers: np.ndarray, fx, fy, cx, cy, c2w: np.ndarray, + res: int, tol: float = TOL): + """Per-voxel visibility from the object's OWN voxels (no depth image). + + Self-occlusion test: every voxel is splatted into a per-view z-buffer over + its projected footprint (min planar depth per pixel); a voxel is visible + iff its own depth is within ``tol`` of the buffer at its centre pixel — + i.e. nothing closer covers it. This is the same decision rule as + ``project_visible_np`` with the rendered depth map replaced by the depth + of the voxelised surface itself (exact up to the 64^3 voxel size, which is + below ``tol`` = 0.02 ~ 1.3 voxels). Returns (u, v, z, visible) like + ``project_visible_np``. Production renders (slat50k cond views) ship no + depth pass, so this is the visibility source for training; the eval sets + keep their rendered GT depth.""" + w2c = np.linalg.inv(c2w) + R, t = w2c[:3, :3], w2c[:3, 3] + xc = centers @ R.T + t + z = xc[:, 2] + u = fx * xc[:, 0] / z + cx + vv = fy * xc[:, 1] / z + cy + ui = np.round(u).astype(int) + vi = np.round(vv).astype(int) + inframe = (z > 0) & (ui >= 0) & (ui < res) & (vi >= 0) & (vi < res) + visible = np.zeros(len(centers), dtype=bool) + if not inframe.any(): + return u, vv, z, visible + # projected voxel footprint (pixels): voxel edge 1/64 at depth z + zf = z[inframe] + r = int(np.clip(np.ceil(0.5 * fx / VOX / float(np.median(zf))), 1, 16)) + buf = np.full((res, res), np.inf, dtype=np.float64) + ui_f, vi_f = ui[inframe], vi[inframe] + for dy in range(-r, r + 1): + yy = vi_f + dy + oky = (yy >= 0) & (yy < res) + for dx in range(-r, r + 1): + xx = ui_f + dx + ok = oky & (xx >= 0) & (xx < res) + np.minimum.at(buf, (yy[ok], xx[ok]), zf[ok]) + dep = buf[vi_f, ui_f] + visible[inframe] = np.abs(zf - dep) < tol + return u, vv, z, visible + + +def raymarch_visible(centers: np.ndarray, coords3: np.ndarray, fx, fy, cx, cy, + c2w: np.ndarray, res: int, skip_vox: float = 2.5, + step_vox: float = 0.5): + """Per-voxel visibility by EXACT ray marching through the 64^3 occupancy + grid (no image-space splatting, hence none of the z-buffer's near-bias): + a voxel is visible iff the ray from its centre to the camera centre meets + no occupied voxel beyond ``skip_vox`` voxels from itself (the skip keeps + the voxel's own surface neighbours from blocking grazing rays, and matches + the rendered-depth reference's TOL=0.02 depth slack; see VIS_SKIP_VOX for + the calibration). Returns (u, v, z, visible) like ``project_visible_np``.""" + occ = np.zeros((VOX, VOX, VOX), dtype=bool) + c3 = np.asarray(coords3, dtype=np.int64) + occ[c3[:, 0], c3[:, 1], c3[:, 2]] = True + w2c = np.linalg.inv(c2w) + R, t = w2c[:3, :3], w2c[:3, 3] + xc = centers @ R.T + t + z = xc[:, 2] + u = fx * xc[:, 0] / z + cx + vv = fy * xc[:, 1] / z + cy + ui = np.round(u).astype(int) + vi = np.round(vv).astype(int) + inframe = (z > 0) & (ui >= 0) & (ui < res) & (vi >= 0) & (vi < res) + cam = c2w[:3, 3] + d = cam[None, :] - centers # (N,3) towards the camera + dist = np.linalg.norm(d, axis=1) + d = d / np.maximum(dist[:, None], 1e-9) + # march only while still inside the cube: t_max = exit distance of the ray + # from [-0.5,0.5]^3 (voxels outside are never occupied) + with np.errstate(divide="ignore", invalid="ignore"): + t1 = (-0.5 - centers) / d + t2 = (0.5 - centers) / d + t_exit = np.nanmin(np.where(d != 0, np.maximum(t1, t2), np.inf), axis=1) + t_max = np.minimum(dist, t_exit) + # march the still-unblocked voxels only (active set shrinks fast: most + # voxels are occluded within a few steps), 16 steps per vectorised chunk + occf = occ.reshape(-1) + step = step_vox / VOX + t = skip_vox / VOX + blocked = np.zeros(len(centers), dtype=bool) + act = np.flatnonzero(inframe) + CH = 16 + while act.size: + tt = t + step * np.arange(CH) # (CH,) + ca, da, tm = centers[act], d[act], t_max[act] + pts = ca[:, None, :] + tt[None, :, None] * da[:, None, :] # (A,CH,3) + valid = tt[None, :] <= tm[:, None] + gi = np.floor((pts + 0.5) * VOX).astype(np.int64) + inside = valid & np.all((gi >= 0) & (gi < VOX), axis=2) + np.clip(gi, 0, VOX - 1, out=gi) + flat = (gi[..., 0] * VOX + gi[..., 1]) * VOX + gi[..., 2] + hit = (occf[flat] & inside).any(axis=1) + blocked[act[hit]] = True + t = float(tt[-1]) + step + act = act[~hit & (tm >= t)] + visible = inframe & ~blocked + return u, vv, z, visible + + +def crop_uv_norm(u: np.ndarray, v: np.ndarray, bbox) -> np.ndarray: + """sample_feats' grid_sample coords (:267-275): render-frame (u,v) -> the + crop-frame [-1,1] convention, float64 -> float32. (N,2).""" + y0, y1, x0, x1 = [float(b) for b in bbox] + # PER-AXIS normalisation: the stored crop is rgba[y0:y1, x0:x1] (no square + # padding, adapt_slat50k.crop_bbox clamps at the frame edge) and is resized + # anisotropically to 518x518, so u maps by the crop WIDTH and v by its + # HEIGHT. Identical to the old single-`side` formula whenever w == h. + w = x1 - x0 + h = y1 - y0 + un = (u - x0 + 0.5) / w * 2 - 1 + vn = (v - y0 + 0.5) / h * 2 - 1 + return np.stack([un, vn], -1).astype(np.float32) + + +def rgba_to_crop(rgba_uint8: np.ndarray) -> Dict[str, np.ndarray]: + """In-memory RGBA uint8 (H,W,4) -> {'premult' [3,518,518] float32 + (dino_features recipe: 518 LANCZOS + alpha-premult on black), + 'rgba_uint8' (H,W,4)}. Shared by the clean-RGBA file loader and the + RecGen matting path (which has no on-disk RGBA).""" + im = Image.fromarray(np.ascontiguousarray(rgba_uint8), mode="RGBA") + im518 = im.resize((518, 518), Image.Resampling.LANCZOS) + a = np.array(im518).astype(np.float32) / 255.0 + rgb = a[:, :, :3] * a[:, :, 3:4] # premult alpha on BLACK + return {"premult": rgb.transpose(2, 0, 1), "rgba_uint8": rgba_uint8} + + +def load_crop_rgba(path: str) -> Dict[str, np.ndarray]: + """RGBA crop file -> {'premult' [3,518,518] float32 (dino_features recipe: + 518 LANCZOS + alpha-premult on black), 'rgba_uint8' (H,W,4)} .""" + return rgba_to_crop(np.array(Image.open(path).convert("RGBA"))) + + +def matte_rgba(rgb_uint8: np.ndarray, mask_uint8: np.ndarray) -> np.ndarray: + """RGB (H,W,3) + binary/greyscale mask (H,W) -> RGBA (H,W,4) uint8 CROPPED + to the mask's bounding box. The RecGen ALPHA the SLAT loader's visibility + gate needs: cond_image has BACKGROUND baked in and no alpha, so the object + mask (prefer cond_mask_sam2, fallback cond_mask) becomes the alpha channel. + Returns the tight-bbox crop so ``crop_uv_norm`` / the alpha gate work exactly + as they do on the clean transparent-background renders.""" + rgb = np.asarray(rgb_uint8) + m = np.asarray(mask_uint8) + if m.ndim == 3: + m = m[..., 0] + a = (m > 127).astype(np.uint8) * 255 + ys, xs = np.nonzero(a) + if len(ys) == 0: # empty mask: whole frame + y0, y1, x0, x1 = 0, rgb.shape[0], 0, rgb.shape[1] + else: + y0, y1 = int(ys.min()), int(ys.max()) + 1 + x0, x1 = int(xs.min()), int(xs.max()) + 1 + rgba = np.concatenate([rgb, a[..., None]], axis=2) # (H,W,4) + crop = rgba[y0:y1, x0:x1] + return crop, (float(y0), float(y1), float(x0), float(x1)) + + +# --------------------------------------------------------------------------- # +class TrellisSlatFlowDataset(TrellisSSFlowDataset): + """SLAT-flow training samples: SS dataset + per-view RGB/DINO inputs, + cameras, and the x1 target hookup.""" + + def __init__(self, *args, toys4k1k_root: str = TOYS4K1K, + max_num_voxels: Optional[int] = MAX_NUM_VOXELS_DEFAULT, + p_seed_none: float = 0.0, # unified mixture: P(EMPTY seed) + p_seed_single: float = 0.0, # P(single-view seed); rest=multi + recgen_roots: Optional[Sequence[str]] = None, + recgen_pose_store: Optional[str] = None, + **kwargs): + super().__init__(*args, **kwargs) + self.toys4k1k_root = toys4k1k_root + # encoded_store recgen root holding the canonical cube cameras + # (pose_cube.npz / views.npz) looked up by sha; None -> analytic fallback. + self._recgen_pose_store = recgen_pose_store + # ---- SS-STAGE seed MIXTURE (mirror of ss_imgcond_dataset.seed_mode) -- + # p_seed_none = P(EMPTY appearance seed -> x0 all-noise, no forcing) + # p_seed_single= P(single-view seed -> n_sub == 1) + # remainder = multi (n_sub ~ U{max(2,min_views)..max_views}) + # BOTH == 0.0 -> the LEGACY path (n_sub ~ U{min_views..max_views}), which + # is byte-identical to the pre-mixture behaviour. VAL (fixed_num_views + # set) is never randomised. Own rng stream (offset 4242 for the mode, + # 7717 for the multi view-count) so the mixture never perturbs the + # existing view-subset / crop draws on the main [seed,epoch,idx] stream. + self.p_seed_none = float(p_seed_none) + self.p_seed_single = float(p_seed_single) + assert self.p_seed_none + self.p_seed_single <= 1.0 + 1e-9, ( + self.p_seed_none, self.p_seed_single) + # TRELLIS-style max_num_voxels FILTER (mirror of datasets/ + # structured_latent.py filter_metadata: metadata['num_voxels'] <= cap): + # applied at index-build time so oversized objects never reach the + # loader. None / 0 disables the filter. + self.max_num_voxels = int(max_num_voxels) if max_num_voxels else None + if self.max_num_voxels: + n_before = len(self.dirs) + self.dirs = [ + d for d in self.dirs + if num_active_voxels(os.path.join(d, "geometry", "voxels.npz")) + <= self.max_num_voxels + ] + n_drop = n_before - len(self.dirs) + print(f"[slatflow-data] split={self.split} max_num_voxels=" + f"{self.max_num_voxels}: kept {len(self.dirs)}/{n_before} " + f"objects (dropped {n_drop})", flush=True) + if not self.dirs: + raise RuntimeError( + f"max_num_voxels={self.max_num_voxels} dropped every object " + f"in split={self.split}") + + # ---- RecGen roots (ADDITIVE, opt-in) -------------------------------- + # RecGen sample objects (// with slat_coords.npy + slat_feats.npy + # + cond_image_*.jpg + cond_mask*.png + view_metadata.json) do NOT carry a + # meta.json / geometry/voxels.npz, so they are invisible to scan_objects. + # Scan them separately and APPEND to self.dirs, tracking which dirs are + # recgen in self._recgen so __getitem__ can route them to the matting / + # released-SLAT loader. Non-recgen behaviour is untouched. + self._recgen = set() + if recgen_roots: + import glob as _glob + if isinstance(recgen_roots, (str, os.PathLike)): + recgen_roots = [recgen_roots] + rg = [] + for root in recgen_roots: + for sc in _glob.iglob(os.path.join(root, "*", "slat_coords.npy")): + rg.append(os.path.dirname(sc)) + rg = sorted(set(rg)) + self._recgen = set(rg) + self.dirs = list(self.dirs) + rg + print(f"[slatflow-data] split={self.split} recgen_roots: added " + f"{len(rg)} recgen objects (total {len(self.dirs)})", flush=True) + + # ---- SS-stage seed mixture (mirror of ss_imgcond_dataset.seed_mode) -- # + def seed_mode(self, idx: int, epoch: int) -> str: + """'none' | 'single' | 'multi' | 'legacy' for this (object, epoch). + + Own rng stream (offset 4242) so the mode decision never perturbs the + [seed,epoch,idx] view-subset / crop draws. VAL (fixed_num_views set) + keeps the fixed eval view count (returns 'single' as an unused label — + __getitem__ pins n_sub to fixed_num_views in that case). With BOTH + knobs 0.0 the mixture is OFF -> 'legacy' (no draw, main stream + untouched).""" + if self.fixed_num_views is not None: # val: unchanged + return "single" + if self.p_seed_none <= 0.0 and self.p_seed_single <= 0.0: + return "legacy" + r = float(np.random.default_rng( + [self.seed, int(epoch), int(idx), 4242]).random()) + if r < self.p_seed_none: + return "none" + if r < self.p_seed_none + self.p_seed_single: + return "single" + return "multi" + + # ---- per-object RGB/bbox adapter ---------------------------------- # + def _rgb_assets(self, obj_dir: str, name: str, n_views: int) -> List[Dict]: + """-> per-view dict(png=..., bbox=(4,), res=int). Production layout + first (views/rgb + views/bbox.npy); toys4k1k fallback for smoke.""" + rgb_dir = os.path.join(obj_dir, "views", "rgb") + bbox_path = os.path.join(obj_dir, "views", "bbox.npy") + if os.path.isdir(rgb_dir) and os.path.isfile(bbox_path): + bboxes = np.load(bbox_path) + out = [] + for i in range(n_views): + png = os.path.join(rgb_dir, f"{i:03d}.png") # lossless (adapter) + if not os.path.isfile(png): + png = os.path.join(rgb_dir, f"{i:03d}.webp") + out.append(dict(png=png, bbox=bboxes[i].astype(np.float64), res=None)) + return out + # toys4k smoke fallback (bbox/res from the render npz; png = the crop) + out = [] + for i in range(n_views): + tag = TOYS4K_VIEW_ORDER[i] + png = os.path.join(self.toys4k1k_root, "inputs", f"{name}_{tag}.png") + rz = np.load(os.path.join(self.toys4k1k_root, "renders", + f"{name}_{tag}.npz")) + out.append(dict(png=png, bbox=rz["bbox"].astype(np.float64), + res=int(rz["res"]))) + return out + + # ---- SS-stage seed-mixture view-count selection --------------------- # + def _mixture_n_sub(self, idx: int, V: int, rng) -> tuple: + """Return (n_sub, no_seed, mode). With the mixture OFF (both knobs 0.0) this + is the LEGACY draw — ``rng.integers(min_views, max_views+1)`` off the + MAIN [seed,epoch,idx] stream — byte-identical to the pre-mixture path. + With the mixture ON the mode comes from the independent 4242 stream + (seed_mode), the multi view-count from the independent 7717 stream, so + the main stream (and thus the ``sub`` view choice) is only advanced by + the legacy branch — exactly as before.""" + if self.fixed_num_views is not None: # VAL: never randomised + return min(int(self.fixed_num_views), V), False, "val" + mode = self.seed_mode(idx, self._epoch) + if mode == "legacy": + return (min(int(rng.integers(self.min_views, self.max_views + 1)), V), + False, "legacy") + if mode == "none": # EMPTY appearance seed: 1 cond view, x0 all-noise + return min(1, V), True, "none" + if mode == "single": + return min(1, V), False, "single" + # multi: n_sub ~ U{max(2,min_views) .. max_views} + lo = max(2, int(self.min_views)) + hi = max(lo, int(self.max_views)) + k = int(np.random.default_rng( + [self.seed, int(self._epoch), int(idx), 7717]).integers(lo, hi + 1)) + return min(k, V), False, "multi" + + # ---- item ----------------------------------------------------------- # + def __getitem__(self, idx: int) -> Dict: + d = self.dirs[idx] + if d in self._recgen: # RecGen loader (additive) + return self._getitem_recgen(idx) + name = os.path.basename(d) + views_path = os.path.join(d, "views", "views.npz") + + # per-(sample, epoch) deterministic RNG — same recipe as SSFlow + rng = np.random.default_rng([self.seed, self._epoch, idx]) + views_all = load_views_npz(views_path) # K, c2w, depth per view + V = len(views_all) + n_sub, no_seed, mode = self._mixture_n_sub(idx, V, rng) + sub = rng.choice(V, size=n_sub, replace=False) + order = list(sub) + [i for i in range(V) if i not in set(sub.tolist())] + + # SLAT coord set (exact inference coordinate ops) + vox = np.load(os.path.join(d, "geometry", "voxels.npz")) + coords = slat_coords_from_voxels(vox["coords"]) # (N,4) int32 + centers = (coords[:, 1:].numpy().astype(np.float64) + 0.5) / VOX - 0.5 + + assets = self._rgb_assets(d, name, V) + + # Only the SAMPLED subset S feeds the model (cond = anchor + S[1:], + # x0 seed = visible-mean over S, x1 = the precomputed official target), + # so crops/uv/vis are built for S ONLY: an 8-view object costs |S| <= 4 + # DINO forwards + projections, not 8. view_order still lists all views. + crops, uvs, viss = [], [], [] + for i in order[:n_sub]: + vw = views_all[i] + a = assets[i] + if a["res"] is not None: + res = a["res"] + elif vw["depth"] is not None: + res = vw["depth"].shape[0] + else: + res = vw["res"] + if vw["depth"] is not None: # rendered GT depth (eval sets) + u, vv, z, vis = project_visible_np( + centers, vw["depth"], vw["fx"], vw["fy"], vw["cx"], vw["cy"], + vw["c2w"], res) + elif VIS_METHOD == "zbuffer": # production: from the voxels + u, vv, z, vis = zbuffer_visible( + centers, vw["fx"], vw["fy"], vw["cx"], vw["cy"], vw["c2w"], res, + tol=VIS_ZBUF_TOL) + else: # production: exact grid ray march + u, vv, z, vis = raymarch_visible( + centers, vox["coords"], vw["fx"], vw["fy"], vw["cx"], vw["cy"], + vw["c2w"], res, skip_vox=VIS_SKIP_VOX) + crop = load_crop_rgba(a["png"]) + uvn = crop_uv_norm(u, vv, a["bbox"]) + # --- MODAL-MASK (alpha) VISIBILITY GATE ------------------------ # + # The geometric tests above (rendered depth / z-buffer / ray march) + # only know about the OBJECT's own surface, so they call a voxel + # visible even when the stored view actually shows something else + # in front of it (PSL occlusion crops) or when the projection falls + # outside the stored crop. The stored RGBA's alpha channel IS the + # per-view MODAL mask (what is really seen), so gate `vis` by it: + # nearest-pixel lookup in the crop frame (grid_sample's + # align_corners=False convention, i.e. px = (un+1)/2*W - 0.5), NO + # dilation. On occluder-free renders (the base pool) alpha covers + # the whole silhouette, so this is a near no-op (edge pixels only). + alpha = crop["rgba_uint8"][..., 3] + Hc, Wc = alpha.shape[0], alpha.shape[1] + px = np.rint((uvn[:, 0].astype(np.float64) + 1.0) * 0.5 * Wc - 0.5).astype(np.int64) + py = np.rint((uvn[:, 1].astype(np.float64) + 1.0) * 0.5 * Hc - 0.5).astype(np.int64) + inside = (px >= 0) & (px < Wc) & (py >= 0) & (py < Hc) + vis = np.asarray(vis) & inside + if inside.any(): + vis[inside] &= alpha[py[inside], px[inside]] > 0 + # --- EMPTY-SEED ('none') gate: contribute ZERO forced x0 rows --- + # The x0 appearance seed is built model-side as the visible-mean of + # the rows any view marks visible; an all-False `vis` therefore + # forces no rows (x0 stays pure noise) while leaving the condition + # (DINO crops / cameras) untouched. Belt-and-suspenders with the + # `no_seed` batch flag honoured in slatflow_model.flow_step. + if no_seed: + vis = np.zeros_like(np.asarray(vis), dtype=bool) + # --------------------------------------------------------------- # + crops.append(crop) + uvs.append(uvn) + viss.append(vis) + + # anchor condition inputs (deployed slat_preprocessor recipe, OQ-6) + slat_input = preprocess_slat_image(crops[0]["rgba_uint8"]) + + # non-anchor Plücker + soft-mask camera dicts (subset views S[1:]) + anchor_c2w = torch.from_numpy(views_all[order[0]]["c2w"]) + pluckers, bias_views = [], [] + for k in range(1, n_sub): + i = order[k] + vw = views_all[i] + a = assets[i] + res = (a["res"] if a["res"] is not None else + (vw["depth"].shape[0] if vw["depth"] is not None else vw["res"])) + pluckers.append(plucker_rays_for_view( + anchor_c2w, torch.from_numpy(vw["c2w"]), + vw["fx"], vw["fy"], vw["cx"], vw["cy"], a["bbox"])) + bias_views.append(dict( + w2c=torch.from_numpy(np.linalg.inv(vw["c2w"])).float(), + fx=vw["fx"], fy=vw["fy"], cx=vw["cx"], cy=vw["cy"], + bbox=[float(b) for b in a["bbox"]], res=float(res))) + plucker = (torch.stack(pluckers) if pluckers + else torch.zeros(0, 1369, 6)) + + item = { + "name": name, + "coords": coords, # (N,4) int32 + "slat_input": slat_input, # dict of 4 CPU tensors + "crops": torch.from_numpy(np.stack([c["premult"] for c in crops])), + "uv": torch.from_numpy(np.stack(uvs)), # (V,N,2) f32 + "vis": torch.from_numpy(np.stack(viss)), # (V,N) bool + "plucker": plucker, # (n_sub-1,1369,6) + "bias_views": bias_views, + "n_subset": n_sub, + "view_order": [int(i) for i in order], + # SS-stage seed-mixture flags (default: legacy/no-op). no_seed=True + # => EMPTY appearance seed (x0 pure noise, honoured in flow_step). + "no_seed": torch.tensor(bool(no_seed)), + "seed_mode": torch.tensor(float(_SEED_MODE_CODE[mode])), + } + + # OFFICIAL precomputed x1 target (tools/gen_x1_official.py output): + # the exact TRELLIS-toolkit latent (SAM3D slat_encoder) on the FULL + # active-voxel set — coord-hash-match its rows onto coords_slat (a + # prune-subset of the voxels) and assert every row is covered. + slat_npz = os.path.join(d, "slat", "slat_official_sam3d.npz") + if os.path.isfile(slat_npz): + z = np.load(slat_npz) + oc = z["coords"].astype(np.int64) # (M,3) all voxels + ok = (oc[:, 0] * VOX + oc[:, 1]) * VOX + oc[:, 2] + order = np.argsort(ok) + want_c = coords[:, 1:].numpy().astype(np.int64) + want = (want_c[:, 0] * VOX + want_c[:, 1]) * VOX + want_c[:, 2] + pos = np.searchsorted(ok[order], want) + sel = order[np.clip(pos, 0, len(order) - 1)] + assert len(ok) > 0 and np.array_equal(ok[sel], want), \ + (f"{name}: slat_official_sam3d.npz does not cover coords_slat " + f"(official voxel set mismatch — regenerate with " + f"tools/gen_x1_official.py)") + item["x1_feats_raw"] = torch.from_numpy( + z["feats"][sel].astype(np.float32)) # (N,8) RAW + return item + + # ---- RecGen matting loader (additive) ------------------------------- # + def _recgen_cube_c2w(self, d: str, n: int): + """Locate the CORRECT camera->cube poses (c2w in the [-0.5,0.5] voxel + frame the SLAT loader projects into). RecGen's cube cameras come from + the ANALYTIC ``pose_cube`` conversion (pose_scale / orientation), NOT a + simple inv(model2world)@cam2world of the raw view_metadata — that lands + the object BEHIND the camera (z<0). So prefer, in order: + 1. ``pose_cube.npz`` (c2w_cube) inside the object dir or a views20/; + 2. the encoded_store's pose_cube.npz / views.npz(c2w_cv), looked up by + sha under ``recgen_pose_store`` when set. + Returns (c2w array (n,4,4)) or None if no canonical source is found.""" + import glob as _glob + cands = [os.path.join(d, "pose_cube.npz"), + os.path.join(d, "views20", "pose_cube.npz"), + os.path.join(d, "views", "pose_cube.npz")] + store = getattr(self, "_recgen_pose_store", None) + if store: + sha = os.path.basename(d) + sf = os.path.join(d, "sha256.txt") + if os.path.isfile(sf): + sha = open(sf).read().strip() or sha + base = os.path.join(store, "objects", sha[:2], sha) + cands += [os.path.join(base, "views20", "pose_cube.npz"), + os.path.join(base, "views", "pose_cube.npz")] + for vp in (os.path.join(base, "views20", "views.npz"), + os.path.join(base, "views", "views.npz")): + if os.path.isfile(vp): + z = np.load(vp) + if "c2w_cv" in z.files: + c = np.asarray(z["c2w_cv"], dtype=np.float64) + if c.shape[0] >= n: + return c[:n] + for p in cands: + if os.path.isfile(p): + z = np.load(p) + key = "c2w_cube" if "c2w_cube" in z.files else ( + "c2w_cv" if "c2w_cv" in z.files else None) + if key is not None: + c = np.asarray(z[key], dtype=np.float64) + if c.shape[0] >= n: + return c[:n] + return None + + def _recgen_cameras(self, d: str): + """RecGen view_metadata.json -> (views_all, V) in the SAME per-view dict + shape ``load_views_npz`` produces (depth=None, fx/fy/cx/cy, c2w, res). + + Intrinsics come from view_metadata.json[i]["intrinsics"] (fx=fy=610, + cx=320, cy=240, 640x480). The camera->cube EXTRINSIC prefers the + canonical pose_cube (see _recgen_cube_c2w); if none is found it falls + back to the ANALYTIC inv(model2world)@cam2world, which is format-correct + but geometrically APPROXIMATE (raw view_metadata alone does not yield the + cube frame — see report caveat #2). transforms.json is empty ({}) and + pose_data.json holds the OBJECT 6D pose, not cameras, in these samples.""" + import json + vm_path = os.path.join(d, "view_metadata.json") + vm = json.load(open(vm_path)) + vm = [json.loads(s) if isinstance(s, str) else s for s in vm] + V = len(vm) + cube = self._recgen_cube_c2w(d, V) # (V,4,4) canonical, or None + views_all = [] + for i, m in enumerate(vm): + K = np.asarray(m["intrinsics"], dtype=np.float64) + H = int(m.get("height", 480)); W = int(m.get("width", 640)) + if cube is not None: + c2w = cube[i] + else: # approximate fallback + c2w = (np.linalg.inv(np.asarray(m["model2world"], dtype=np.float64)) + @ np.asarray(m["cam2world"], dtype=np.float64)) + views_all.append(dict( + depth=None, + fx=float(K[0, 0]), fy=float(K[1, 1]), + cx=float(K[0, 2]), cy=float(K[1, 2]), + c2w=c2w, res=int(max(H, W)))) + return views_all, V + + def _recgen_asset(self, d: str, i: int): + """View i -> {'rgba': matted RGBA crop, 'bbox': (y0,y1,x0,x1), 'res': int}. + cond_image_i (RGB, background baked in) x cond_mask (prefer the SAM2 mask, + fallback the plain mask) -> RGBA so the SLAT alpha-visibility gate works.""" + img_p = os.path.join(d, f"cond_image_{i:02d}.jpg") + rgb = np.array(Image.open(img_p).convert("RGB")) + mask_p = os.path.join(d, f"cond_mask_sam2_{i:02d}.png") + if not os.path.isfile(mask_p): + mask_p = os.path.join(d, f"cond_mask_{i:02d}.png") + mask = np.array(Image.open(mask_p).convert("L")) + crop_rgba, bbox = matte_rgba(rgb, mask) + return dict(rgba=crop_rgba, bbox=bbox, res=int(max(rgb.shape[0], rgb.shape[1]))) + + def _getitem_recgen(self, idx: int) -> Dict: + d = self.dirs[idx] + name = os.path.basename(d) + + rng = np.random.default_rng([self.seed, self._epoch, idx]) + views_all, V = self._recgen_cameras(d) + n_sub, no_seed, mode = self._mixture_n_sub(idx, V, rng) + sub = rng.choice(V, size=n_sub, replace=False) + order = list(sub) + [i for i in range(V) if i not in set(sub.tolist())] + + # x1 = RecGen's RELEASED SLAT (slat_coords 0..63 int + slat_feats [N,8]). + # The released coords ARE the final SLAT coord set (1:1 with slat_feats), + # so they are used directly as `coords` (no prune/downsample: RecGen ships + # no voxels.npz) and slat_feats is the row-aligned x1 target. Coverage + # is 1:1 by construction; assert the row counts match. + sc = np.load(os.path.join(d, "slat_coords.npy")).astype(np.int64) # (N,3) + feats = np.load(os.path.join(d, "slat_feats.npy")).astype(np.float32) # (N,8) + assert sc.ndim == 2 and sc.shape[1] == 3 and feats.shape[0] == sc.shape[0], \ + (f"{name}: recgen slat_coords {sc.shape} / slat_feats {feats.shape} " + f"mismatch (x1 rows must align 1:1 with coords)") + assert sc.min() >= 0 and sc.max() < VOX, (name, sc.min(), sc.max()) + coords = torch.cat([torch.zeros(len(sc), 1, dtype=torch.int32), + torch.from_numpy(sc).int()], dim=1) # (N,4) int32 + centers = (sc.astype(np.float64) + 0.5) / VOX - 0.5 + + crops, uvs, viss = [], [], [] + for i in order[:n_sub]: + vw = views_all[i] + a = self._recgen_asset(d, i) + res = a["res"] + u, vv, z, vis = raymarch_visible( + centers, sc, vw["fx"], vw["fy"], vw["cx"], vw["cy"], + vw["c2w"], res, skip_vox=VIS_SKIP_VOX) + crop = rgba_to_crop(a["rgba"]) + uvn = crop_uv_norm(u, vv, a["bbox"]) + # modal-mask (alpha) gate — identical to the clean-RGBA path + alpha = crop["rgba_uint8"][..., 3] + Hc, Wc = alpha.shape[0], alpha.shape[1] + px = np.rint((uvn[:, 0].astype(np.float64) + 1.0) * 0.5 * Wc - 0.5).astype(np.int64) + py = np.rint((uvn[:, 1].astype(np.float64) + 1.0) * 0.5 * Hc - 0.5).astype(np.int64) + inside = (px >= 0) & (px < Wc) & (py >= 0) & (py < Hc) + vis = np.asarray(vis) & inside + if inside.any(): + vis[inside] &= alpha[py[inside], px[inside]] > 0 + if no_seed: + vis = np.zeros_like(np.asarray(vis), dtype=bool) + crops.append(crop) + uvs.append(uvn) + viss.append(vis) + a["_bbox"] = a["bbox"] # keep for plucker reuse + views_all[i]["_asset"] = a + + slat_input = preprocess_slat_image(crops[0]["rgba_uint8"]) + + anchor_c2w = torch.from_numpy(views_all[order[0]]["c2w"]) + pluckers, bias_views = [], [] + for k in range(1, n_sub): + i = order[k] + vw = views_all[i] + a = vw["_asset"] + res = a["res"] + pluckers.append(plucker_rays_for_view( + anchor_c2w, torch.from_numpy(vw["c2w"]), + vw["fx"], vw["fy"], vw["cx"], vw["cy"], a["bbox"])) + bias_views.append(dict( + w2c=torch.from_numpy(np.linalg.inv(vw["c2w"])).float(), + fx=vw["fx"], fy=vw["fy"], cx=vw["cx"], cy=vw["cy"], + bbox=[float(b) for b in a["bbox"]], res=float(res))) + plucker = (torch.stack(pluckers) if pluckers else torch.zeros(0, 1369, 6)) + + return { + "name": name, + "coords": coords, + "slat_input": slat_input, + "crops": torch.from_numpy(np.stack([c["premult"] for c in crops])), + "uv": torch.from_numpy(np.stack(uvs)), + "vis": torch.from_numpy(np.stack(viss)), + "plucker": plucker, + "bias_views": bias_views, + "n_subset": n_sub, + "view_order": [int(i) for i in order], + "no_seed": torch.tensor(bool(no_seed)), + "seed_mode": torch.tensor(float(_SEED_MODE_CODE[mode])), + "x1_feats_raw": torch.from_numpy(feats), # (N,8) RAW released SLAT + } + + +def collate_single(batch: Sequence[Dict]) -> Dict: + """bs=1 collate (variable N per object => per-object steps + grad accum).""" + assert len(batch) == 1, \ + "SLAT flow trains bs=1 per object (grad accum for larger batches)" + return batch[0] + + +def collate_batched(batch: Sequence[Dict]) -> Dict: + """Multi-object collate mirroring TRELLIS ``SLat.collate_fn``: the B + objects' SLAT coords are CONCATENATED into one coordinate set whose batch + column ([b,x,y,z]) is the object index, with per-object ``layout`` slices + recorded. Everything with a per-object variable shape (crops, uv, vis, + plucker, condition inputs, targets) stays a python list — the batched + model path builds each object's own condition and seed from them. + + Consumed by ``SlatFlowModel.sample_batch`` (true batched inference: + one forward per solver step for all B objects).""" + coords_parts: List[torch.Tensor] = [] + layout: List[slice] = [] + ofs = 0 + for i, it in enumerate(batch): + c = it["coords"].clone() + assert int(c[:, 0].max()) == 0, "per-object coords must have batch col 0" + c[:, 0] = i # batch-index column + coords_parts.append(c) + layout.append(slice(ofs, ofs + c.shape[0])) + ofs += c.shape[0] + return { + "coords": torch.cat(coords_parts, dim=0), # (T_total,4) int32 + "layout": layout, # per-object slices + "names": [it["name"] for it in batch], + "slat_input": [it["slat_input"] for it in batch], + "crops": [it["crops"] for it in batch], + "uv": [it["uv"] for it in batch], + "vis": [it["vis"] for it in batch], + "plucker": [it["plucker"] for it in batch], + "bias_views": [it["bias_views"] for it in batch], + "n_subset": [int(it["n_subset"]) for it in batch], + "view_order": [it["view_order"] for it in batch], + "x1_feats_raw": [it.get("x1_feats_raw") for it in batch], + # SS-stage seed-mixture flags (per object; default legacy/no-op) + "no_seed": [bool(it["no_seed"]) if it.get("no_seed") is not None else False + for it in batch], + "seed_mode": [float(it["seed_mode"]) if it.get("seed_mode") is not None else 3.0 + for it in batch], + } diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/__init__.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/__init__.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..63ddb634e483ffa61deb203480fe096ba0edb4e8 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/__init__.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..45ecb612b58986b75a3d36b0f316ba050854c54d Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/condition.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/condition.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3b0b28283cebd9c24e8ee16cd69f16ea9ed7abb6 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/condition.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/flow_model.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/flow_model.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e72452197595c00cb432e8636ffbb8b2e8b7bd81 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/flow_model.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/generator_hooks.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/generator_hooks.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b888ae0112ea44fef8558551c640a319e926263d Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/generator_hooks.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/mv_slat_condition.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/mv_slat_condition.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..00581880890b192abcd847534fef9f9e6295a0b6 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/mv_slat_condition.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/mv_ss_condition.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/mv_ss_condition.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d23d9d572631598419c1809bc488a7906d48cd6c Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/mv_ss_condition.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/mv_ss_condition.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/mv_ss_condition.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2cb2da8a4ddbeb6df61a09f4369504bf860a3add Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/mv_ss_condition.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/slat_pretrained.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/slat_pretrained.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..415062f8bfbbdd49378afccd232735761685cbee Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/slat_pretrained.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/slat_pretrained.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/slat_pretrained.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7b8baf564facf56b8f7c22e742e8d910d64d2915 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/slat_pretrained.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/slatflow_model.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/slatflow_model.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..75223b9be211e9eb7b97362fdafffb6bd0c68ed0 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/slatflow_model.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/ss_vae.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/ss_vae.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e71bb77201cb60d29fa02b2f3e555651a1d70ef3 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/ss_vae.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/ssflow_model.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/ssflow_model.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..da2c992b0c0b5988457ece8ad4f8a2a530752d85 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/ssflow_model.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/ssflow_model.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/ssflow_model.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6b3f4222c0a69f073f2659403d0edb6f57ec2883 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/model/__pycache__/ssflow_model.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6130d994675a52c0029e819c7151d4701059826b Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/ema.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/ema.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..300f8f5be4aa100502a4c1a6106335186774db74 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/ema.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..da3011de35da7e814202133e42037303ceac1490 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train_slat_flow.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train_slat_flow.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..164fb76af5005f5f36d4dd573e243e6728cd6981 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train_slat_flow.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train_ss_encoder.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train_ss_encoder.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3266a70edbf018f32fb0f96454c049736d92c5aa Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train_ss_encoder.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train_ss_encoder.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train_ss_encoder.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e7ea49d452dd9f1306a2c44550a2c4b40735fd88 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/train_ss_encoder.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/val_appforce.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/val_appforce.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f271653e426cbbc742b0e43e5ec7c11cd0ca4537 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/val_appforce.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/validate.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/validate.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a7e42c895e5d1d3ea1c54ad7f6a51df7e4e37361 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/validate.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/validate.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/validate.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..4f6078c8fd6045a08618e3c944284da4a0c066d3 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/__pycache__/validate.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/train_slat_flow.py b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/train_slat_flow.py index 0e110dfdd436ca8ee9693465cf0354a65c91eb37..4cb4b20e3c87fd6aed803f584c9e99ae6555f107 100644 --- a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/train_slat_flow.py +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/train_slat_flow.py @@ -223,9 +223,9 @@ def build_dataset(data_cfg: dict, split: str): p_seed_single=data_cfg.get("p_seed_single", 0.0), filter_bad_views=data_cfg.get("filter_bad_views", False), min_visfrac=data_cfg.get("min_visfrac", 0.05), - # condition variants (default OFF -> legacy white-cond-only index) + # condition sources: ONE POOL of white cond + RecGen views, anchor's + # source decides all aux views (ON in train_slatflow_prod.yaml) use_recgen_images=data_cfg.get("use_recgen_images", False), - p_recgen_image=data_cfg.get("p_recgen_image", 0.0), cond_cameras_fallback=data_cfg.get("cond_cameras_fallback", False), ) val_roots = data_cfg.get("val_roots") or [] diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/train_slat_flow.py.bak_predatveiw b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/train_slat_flow.py.bak_predatveiw new file mode 100644 index 0000000000000000000000000000000000000000..83e29fabfb4fcd85de79a1f45ab0ee68f08f87f4 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/mvsam3d/train/train_slat_flow.py.bak_predatveiw @@ -0,0 +1,579 @@ +"""Training loop for the multi-view SLAT flow (SlatFlowModel, SLAT_GEN_PLAN §10-5). + +Rectified-flow v-prediction in SAM-3D's slat_mean/std-NORMALIZED latent space +between two DATA distributions: x0 = the forced appearance seed (visible rows = +mean-DINO -> frozen SLatEncoder -> normalized; INVISIBLE rows = fresh N(0,1) +per iteration) and x1 = the full-view visible-mean target SLAT (rows never +visible in any view are MASKED OUT of the loss). Warm-started from the +deployed slat_generator.ckpt; the ONLY architectural additions are the +multi-view condition concat (+ zero-init Plücker FC) and the local soft-mask +cross-attention bias (§3/§4), both of which leave the model exactly pretrained +at N=1 / step 0. + +Structure mirrors train_ss_encoder.py: AdamW + autocast; checkpointing / +scheduling / distributed REUSED VERBATIM from the vendored fork module +``sam3d_objects.training.stage1_flow_lora``. Single-GPU (WORLD_SIZE=1) +transparently skips the process group and DDP wrap. + +The SLAT stage is a dense [1, N, 8] tensor over a PER-OBJECT coord set, so the +loader batch size is pinned to 1 (collate_single) and effective batching is +grad accumulation. + +ALL hyperparameters come from a YAML config (default configs/train_slatflow.yaml): + + SPARSE_ATTN_BACKEND=sdpa PYTHONPATH=. \ + torchrun --nproc_per_node=N mvsam3d/train/train_slat_flow.py \ + --config configs/train_slatflow.yaml --set train.steps=2000 +""" +from __future__ import annotations + +import argparse +import os +import random +import time +from contextlib import nullcontext +from types import SimpleNamespace + +import numpy as np +import torch +import yaml +from torch.utils.data import DataLoader, DistributedSampler + +from sam3d_objects.training.stage1_flow_lora.distributed import ( + barrier, + cleanup_distributed, + init_distributed, + reduce_scalar_dict, + setup_runtime, +) +from sam3d_objects.training.stage1_flow_lora.train_utils import build_scheduler +from sam3d_objects.training.stage1_flow_lora.checkpointing import ( + load_checkpoint, + save_checkpoint, +) +from mvsam3d.data.slatflow_dataset import ( + TrellisSlatFlowDataset, + collate_batched, + collate_single, +) +from mvsam3d.data.ssflow_dataset import scan_objects +from mvsam3d.model.slatflow_model import SlatFlowModel +from mvsam3d.train.ema import EmaShadow, load_ema_sidecar, save_ema_sidecar +from mvsam3d.train import val_appforce +from mvsam3d import paths as _paths + +DEFAULT_CONFIG = os.path.join(_paths.CONFIG_DIR, "train_slatflow.yaml") + +try: + import wandb +except ImportError: # optional dependency + wandb = None + +_WANDB_RUN = None # rank-0 run handle (None = disabled) + + +def init_wandb(cfg: dict, is_main: bool) -> None: + """Start the rank-0 W&B run from cfg["wandb"]; no-op elsewhere/when off.""" + global _WANDB_RUN + wb_cfg = cfg.get("wandb") or {} + if not is_main or not wb_cfg.get("enabled", False) or wandb is None: + return + run_dir = wb_cfg.get("dir") or None + if run_dir: + os.makedirs(run_dir, exist_ok=True) + _WANDB_RUN = wandb.init( + project=wb_cfg.get("project", "mvsam3d-slatflow"), + entity=wb_cfg.get("entity") or None, + name=wb_cfg.get("name") or None, + dir=run_dir, + config=cfg, + ) + print(f"[wandb] {_WANDB_RUN.url}", flush=True) + + +def log_metrics(metrics: dict, step: int) -> None: + """Log any dict of floats at ``step`` (no-op off rank 0 / when W&B is off). + + Module-level hook so out-of-loop evaluators can report too, e.g. + ``log_metrics({"val/chamfer": .., "val/psnr": ..}, step)``.""" + if _WANDB_RUN is None: + return + payload = {k: float(v) for k, v in metrics.items()} + payload["step"] = int(step) + _WANDB_RUN.log(payload, step=int(step)) + + +def finish_wandb() -> None: + global _WANDB_RUN + if _WANDB_RUN is not None: + _WANDB_RUN.finish() + _WANDB_RUN = None + + +def move_to(obj, device): + if torch.is_tensor(obj): + return obj.to(device, non_blocking=True) + if isinstance(obj, dict): + return {k: move_to(v, device) for k, v in obj.items()} + if isinstance(obj, list): + return [move_to(v, device) for v in obj] + return obj + + +def _coerce(text: str): + return yaml.safe_load(text) + + +def _apply_override(cfg: dict, dotted_key: str, value) -> None: + keys = dotted_key.split(".") + node = cfg + for k in keys[:-1]: + if k not in node or not isinstance(node[k], dict): + raise KeyError(f"--set {dotted_key}: '{k}' is not a config section") + node = node[k] + if keys[-1] not in node: + raise KeyError(f"--set {dotted_key}: unknown key '{keys[-1]}'") + node[keys[-1]] = value + + +def load_config(path: str, overrides: list[str]) -> dict: + with open(path) as f: + cfg = yaml.safe_load(f) + cfg = _paths.expand(cfg) + for ov in overrides or []: + if "=" not in ov: + raise ValueError(f"--set expects key=value, got {ov!r}") + key, raw = ov.split("=", 1) + _apply_override(cfg, key.strip(), _coerce(raw.strip())) + return cfg + + +def seed_everything(seed: int, rank: int) -> None: + seed = int(seed) + int(rank) + random.seed(seed) + np.random.seed(seed) + torch.manual_seed(seed) + torch.cuda.manual_seed_all(seed) + + +def build_dataset(data_cfg: dict, split: str): + """Same split policy as the SS trainer: explicit val_roots when non-empty + (train = roots minus val ids), else the legacy deterministic hash split.""" + val_roots = data_cfg.get("val_roots") or [] + common = dict( + min_views=data_cfg["min_views"], + max_views=data_cfg["max_views"], + fixed_num_views=(data_cfg["val_num_views"] if split == "val" else None), + seed=data_cfg["aug_seed"], + max_num_voxels=data_cfg.get("max_num_voxels", 20000), + # SS-stage seed MIXTURE (default 0.0 -> legacy n_sub~U{min..max}); VAL is + # never randomised regardless. RecGen roots are additive/opt-in. + p_seed_none=data_cfg.get("p_seed_none", 0.0), + p_seed_single=data_cfg.get("p_seed_single", 0.0), + recgen_roots=(data_cfg.get("recgen_roots") or None) if split != "val" else None, + recgen_pose_store=data_cfg.get("recgen_pose_store") or None, + ) + if val_roots: + if split == "val": + return TrellisSlatFlowDataset(roots=val_roots, split="all", **common) + val_names = {os.path.basename(d) for d in scan_objects(val_roots)} + return TrellisSlatFlowDataset(roots=data_cfg["roots"], split="all", + exclude_names=val_names, **common) + return TrellisSlatFlowDataset( + roots=data_cfg["roots"], + split=split, + val_fraction=data_cfg["val_fraction"], + split_seed=data_cfg["split_seed"], + **common, + ) + + +@torch.no_grad() +def run_val(model, val_loader, dist_state, device, amp, autocast_dt, + max_batches: int) -> dict: + """Held-out flow v-MSE (same objective as training) under a fixed RNG so + the metric is comparable across passes. Decode/faithfulness eval is the + TrainTest phase — this in-loop val only tracks the flow objective.""" + model.eval() + core = model.module if hasattr(model, "module") else model + cpu_state = torch.get_rng_state() + cuda_state = torch.cuda.get_rng_state(device) + torch.manual_seed(0) # fixes the x0 invisible noise + t + tot, n = {}, 0 + for bi, batch in enumerate(val_loader): + if 0 < max_batches <= bi: + break + batch = move_to(batch, device) + with (torch.autocast("cuda", dtype=autocast_dt) if amp else nullcontext()): + losses = core.flow_step(batch) + for k, v in losses.items(): + tot[k] = tot.get(k, 0.0) + float(v.detach()) + n += 1 + torch.set_rng_state(cpu_state) + torch.cuda.set_rng_state(cuda_state, device) + avg = {k: torch.tensor(v / max(1, n)) for k, v in tot.items()} + avg["val_samples"] = torch.tensor(float(n)) + reduced = reduce_scalar_dict(avg, dist_state) + model.train() + return reduced + + + +def _raise_nccl_timeout(dist_state, hours: float) -> None: + try: + import datetime + import torch.distributed as dist + pg = dist.distributed_c10d._get_default_group() + backend = pg._get_backend(dist_state.device) + backend._set_default_timeout(datetime.timedelta(hours=hours)) + if dist_state.is_main: + print(f"[setup] NCCL default timeout -> {hours} h", flush=True) + except Exception as e: # best effort; older torch + if dist_state.is_main: + print(f"[setup] could not raise NCCL timeout: {e}", flush=True) + + +def _appforce_val(model, ema, ema_eval, af_cfg, step, dist_state, is_main): + """Appearance-forcing validation (decode + LPIPS/SSIM/PSNR/Chamfer on the + toys4k-100 / omni3d eval sets) with the EMA weights swapped in.""" + core = model.module if hasattr(model, "module") else model + use_ema_weights = ema is not None and ema_eval + swap_ctx = (ema.swapped_into(model) if use_ema_weights else nullcontext()) + if is_main: + print(f"[val_af]{' [ema]' if use_ema_weights else ''} step {step}: " + f"starting appearance-forcing validation", flush=True) + with swap_ctx: + val_appforce.run_validation(core, af_cfg, step, dist_state, + run_root=af_cfg["out_root"], + log_fn=log_metrics if is_main else None) + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--config", type=str, default=DEFAULT_CONFIG, + help="Path to the training YAML config.") + ap.add_argument("--set", dest="overrides", action="append", default=[], + metavar="KEY.SUBKEY=VALUE", + help="Override a single config value (repeatable).") + args = ap.parse_args() + + cfg = load_config(args.config, args.overrides) + data_cfg = cfg["data"] + model_cfg = cfg["model"] + optim_cfg = cfg["optim"] + train_cfg = cfg["train"] + ema_cfg = cfg.get("ema") or {} + + # ── distributed init (reused vendored infra) ── + dist_state = init_distributed(train_cfg["backend"]) + if dist_state.distributed: + # the appearance-forcing validation pauses the loop for up to ~1 h per + # pass; lift NCCL's 10-min default collective timeout so the barrier + # around it cannot kill the run. + _raise_nccl_timeout(dist_state, hours=6) + seed_everything(train_cfg["seed"], dist_state.rank) + device = dist_state.device + is_main = dist_state.is_main + if is_main: + os.makedirs(train_cfg["ckpt_dir"], exist_ok=True) + barrier() + + # ── data ── + # batch_objects == 1 (default): loader bs=1 + collate_single, effective + # batch via grad_accum_steps (legacy behaviour). + # batch_objects > 1: TRUE multi-object batching — loader bs=batch_objects + # + collate_batched, ONE backbone forward per optimizer step + # (SlatFlowModel.flow_step_batched); grad_accum_steps MUST be 1. + num_workers = train_cfg["num_workers"] + batch_objects = int(train_cfg.get("batch_objects", 1) or 1) + if batch_objects > 1 and int(train_cfg.get("grad_accum_steps", 1) or 1) != 1: + raise ValueError( + "train.batch_objects > 1 requires train.grad_accum_steps == 1 " + "(true batching replaces accumulation)") + train_collate = collate_batched if batch_objects > 1 else collate_single + train_ds = build_dataset(data_cfg, "train") + train_sampler = ( + DistributedSampler(train_ds, num_replicas=dist_state.world_size, + rank=dist_state.rank, shuffle=True) + if dist_state.distributed else None + ) + if is_main: + print(f"[train] train samples: {len(train_ds)}", flush=True) + train_loader = DataLoader( + train_ds, batch_size=batch_objects, sampler=train_sampler, + shuffle=(train_sampler is None), num_workers=num_workers, + collate_fn=train_collate, drop_last=True, + persistent_workers=num_workers > 0, pin_memory=True, + ) + + # Warm the lazy imports used inside __getitem__ (open3d, hydra, ...) in the + # MAIN process BEFORE wandb.init: wandb 0.29 installs import hooks that + # raise ForkedError when they first fire inside a forked DataLoader worker. + _ = train_ds[0] + init_wandb(cfg, is_main) + + val_every = train_cfg["val_every"] + val_loader = None + has_val = bool(data_cfg.get("val_roots")) or data_cfg["val_fraction"] > 0.0 + if val_every > 0 and has_val: + try: + val_ds = build_dataset(data_cfg, "val") + except RuntimeError: + val_ds = None + if val_ds is not None: + if is_main: + print(f"[train] val samples: {len(val_ds)}", flush=True) + val_sampler = ( + DistributedSampler(val_ds, num_replicas=dist_state.world_size, + rank=dist_state.rank, shuffle=False) + if dist_state.distributed else None + ) + val_loader = DataLoader( + val_ds, batch_size=1, sampler=val_sampler, shuffle=False, + num_workers=max(1, num_workers // 2), collate_fn=collate_single, + drop_last=False, pin_memory=True, + ) + + # ── model + optimizer ── + # accept the two new flags; fall back to legacy enable_bias if present + enable_soft_mask = model_cfg.get("enable_soft_mask", + model_cfg.get("enable_bias", True)) + enable_plucker = model_cfg.get("enable_plucker", + model_cfg.get("enable_bias", True)) + model = SlatFlowModel( + pretrained_ckpt=model_cfg["pretrained_ckpt"] or None, + enable_soft_mask=enable_soft_mask, + enable_plucker=enable_plucker, + bias_value=model_cfg["bias_value"], + fp16_torso=model_cfg["fp16_torso"], + use_checkpoint=model_cfg.get("use_checkpoint", False), + ).to(device) + if is_main and model.use_checkpoint: + print(f"[setup] gradient checkpointing ON " + f"({model.n_checkpointed_blocks} blocks)", flush=True) + # the DINO towers self-freeze (freeze_backbone=True); optionally freeze the + # whole pretrained fuser (projections + slot embeddings) too. + if not model_cfg["train_condition_embedder"]: + model.mv_embedder.fuser.requires_grad_(False) + model.train() + trainable = [p for p in model.parameters() if p.requires_grad] + if is_main: + n_tr = sum(p.numel() for p in trainable) + n_all = sum(p.numel() for p in model.parameters()) + print(f"[setup] trainable params: {n_tr/1e6:.1f}M / {n_all/1e6:.1f}M " + f"(enable_soft_mask={enable_soft_mask}, " + f"enable_plucker={enable_plucker}, " + f"fp16_torso={model_cfg['fp16_torso']})", flush=True) + + optimizer = torch.optim.AdamW( + trainable, lr=optim_cfg["lr"], weight_decay=optim_cfg["weight_decay"], + ) + sched_cfg = SimpleNamespace( + max_steps=train_cfg["steps"], warmup_steps=train_cfg["warmup_steps"], + min_lr=train_cfg["min_lr"], lr=optim_cfg["lr"], + scheduler=train_cfg["scheduler"], + ) + scheduler = build_scheduler(optimizer, sched_cfg) + runtime = setup_runtime( + model=model, optimizer=optimizer, scheduler=scheduler, state=dist_state, + deepspeed_config_path=train_cfg["deepspeed_config"], + ) + model = runtime.model + optimizer = runtime.optimizer + scheduler = runtime.scheduler + trainable = [p for p in model.parameters() if p.requires_grad] + + steps = train_cfg["steps"] + log_every = train_cfg["log_every"] + ckpt_every = train_cfg["ckpt_every"] + ckpt_dir = train_cfg["ckpt_dir"] + grad_clip = optim_cfg["grad_clip"] + grad_accum = max(1, train_cfg["grad_accum_steps"]) + amp = train_cfg["amp"] + amp_dtype = train_cfg["amp_dtype"] + autocast_dt = torch.bfloat16 if str(amp_dtype).lower() == "bf16" else torch.float16 + use_scaler = amp and str(amp_dtype).lower() in {"fp16", "float16"} + scaler = torch.cuda.amp.GradScaler(enabled=use_scaler) + + # ── resume (REUSED fork load_checkpoint) ── + start_step = 0 + resume_from = train_cfg["resume_from"] + if resume_from: + start_step, _ = load_checkpoint( + checkpoint_path=resume_from, model=model, optimizer=optimizer, + scheduler=scheduler, scaler=scaler, map_location="cpu", + ) + if is_main: + print(f"[resume] from {resume_from} at step {start_step}", flush=True) + # The optimizer/scheduler sidecar carries the OLD base lr (param_groups + # 'lr'/'initial_lr' + LambdaLR base_lrs) and would silently override a + # changed config lr. Re-anchor the schedule on the config value. + cfg_lr = float(optim_cfg["lr"]) + lambdas = getattr(scheduler, "lr_lambdas", None) + if lambdas is not None: + scheduler.base_lrs = [cfg_lr] * len(optimizer.param_groups) + for g, f in zip(optimizer.param_groups, lambdas): + g["initial_lr"] = cfg_lr + g["lr"] = cfg_lr * float(f(scheduler.last_epoch)) + if is_main: + print(f"[resume] lr re-anchored to config: base={cfg_lr:.3e} " + f"current={optimizer.param_groups[0]['lr']:.6e}", flush=True) + + # ── EMA (fp32 shadow of the trainable params; sidecar-persisted) ── + ema = None + ema_eval = bool(ema_cfg.get("eval_with_ema", True)) + if bool(ema_cfg.get("enable", False)): + ema = EmaShadow(model, rate=ema_cfg.get("rate", 0.9999)) + loaded = bool(resume_from) and load_ema_sidecar(resume_from, ema, model) + if is_main: + src = "resumed from sidecar" if loaded else "initialized from params" + print(f"[ema] enabled rate={ema.rate} eval_with_ema={ema_eval} " + f"({src}, {len(ema.shadow)} tensors)", flush=True) + + if is_main: + gbs = grad_accum * batch_objects * dist_state.world_size + print(f"[setup] backend={dist_state.backend} world={dist_state.world_size} " + f"global_batch={gbs} (objects/optim-step) steps={steps} " + f"batch_objects={batch_objects} " + f"grad_accum={grad_accum} amp={amp}/{amp_dtype} " + f"val_every={val_every}", flush=True) + + # ── appearance-forcing validation config (val_appforce.py) ── + af_cfg = cfg.get("val_appforce") or {} + af_enabled = bool(af_cfg.get("enabled", False)) + af_every = int(af_cfg.get("every", 0)) if af_enabled else 0 + if af_enabled and bool(af_cfg.get("at_step0", True)) and start_step == 0: + _appforce_val(model, ema, ema_eval, af_cfg, 0, dist_state, is_main) + + # ── training loop ── + optimizer.zero_grad(set_to_none=True) + step, micro, epoch = start_step, 0, 0 + if train_sampler is not None and hasattr(train_sampler, "set_epoch"): + train_sampler.set_epoch(epoch) + train_ds.set_epoch(epoch) + train_iter = iter(train_loader) + accum = {} + t0 = time.time() + ok = False + + try: + while step < steps: + try: + batch = next(train_iter) + except StopIteration: + epoch += 1 + if train_sampler is not None and hasattr(train_sampler, "set_epoch"): + train_sampler.set_epoch(epoch) + train_ds.set_epoch(epoch) # fresh view-subset augmentation + train_iter = iter(train_loader) + batch = next(train_iter) + batch = move_to(batch, device) + + autocast = (torch.autocast("cuda", dtype=autocast_dt) if amp + else nullcontext()) + with autocast: + losses = model(batch) # DDP forward == flow_step + loss = losses["loss"] / grad_accum + + micro += 1 + is_boundary = (micro % grad_accum == 0) + sync_ctx = (nullcontext() + if (is_boundary or not hasattr(model, "no_sync")) + else model.no_sync()) + with sync_ctx: + (scaler.scale(loss) if use_scaler else loss).backward() + + for k, v in losses.items(): + accum[k] = accum.get(k, 0.0) + float(v.detach()) + + if not is_boundary: + continue + + grad_norm = None + if grad_clip > 0: + if use_scaler: + scaler.unscale_(optimizer) + grad_norm = torch.nn.utils.clip_grad_norm_(trainable, grad_clip) + if use_scaler: + scaler.step(optimizer) + scaler.update() + else: + optimizer.step() + if ema is not None: + ema.update(model) # AFTER optimizer.step() + optimizer.zero_grad(set_to_none=True) + scheduler.step() + step += 1 + + if os.environ.get("VRAM_PROBE"): + alloc = torch.cuda.max_memory_allocated(device) / 1e9 + reserv = torch.cuda.max_memory_reserved(device) / 1e9 + print(f"[vram_probe] step {step} max_allocated={alloc:.3f}GB " + f"max_reserved={reserv:.3f}GB", flush=True) + + if step % log_every == 0: + avg = {k: torch.tensor(v / (log_every * grad_accum)) + for k, v in accum.items()} + reduced = reduce_scalar_dict(avg, dist_state) + accum.clear() + if is_main: + rate = log_every / (time.time() - t0) + lr = optimizer.param_groups[0]["lr"] + msg = " ".join(f"{k}={reduced[k]:.4f}" for k in sorted(reduced)) + print(f"[train] step {step} lr={lr:.6e} {msg} " + f"({rate:.2f} it/s)", flush=True) + wb = {f"train/{k}": float(reduced[k]) for k in reduced} + wb["train/lr"] = lr + wb["train/it_per_s"] = rate + if grad_norm is not None: + wb["train/grad_norm"] = float(grad_norm) + log_metrics(wb, step) + t0 = time.time() + + if val_loader is not None and step % val_every == 0: + use_ema_weights = ema is not None and ema_eval + swap_ctx = (ema.swapped_into(model) if use_ema_weights + else nullcontext()) + with swap_ctx: # training weights restored on exit + metrics = run_val(model, val_loader, dist_state, device, amp, + autocast_dt, + max_batches=train_cfg["val_max_batches"]) + if is_main: + tag = " [ema]" if use_ema_weights else "" + msg = " ".join(f"{k}={metrics[k]:.4f}" for k in sorted(metrics) + if k != "val_samples") + print(f"[val]{tag} step {step} {msg} " + f"(n={int(metrics['val_samples'])})", flush=True) + log_metrics({f"val/{k}": float(v) for k, v in metrics.items() + if k != "val_samples"}, step) + + if af_every > 0 and step % af_every == 0: + _appforce_val(model, ema, ema_eval, af_cfg, step, dist_state, is_main) + t0 = time.time() # don't count the val pause as train time + + if step % ckpt_every == 0: + if is_main: + path = f"{ckpt_dir}/step_{step:07d}.pt" + save_checkpoint( + checkpoint_path=path, model=model, optimizer=optimizer, + scheduler=scheduler, scaler=scaler, step=step, + config_dict=cfg, + ) + if ema is not None: + save_ema_sidecar(path, ema, step) + print(f"[train] saved {path}" + f"{' (+ema sidecar)' if ema is not None else ''}", flush=True) + barrier() + + if step >= steps: + break + ok = True + finally: + if is_main: + print(f"[train] {'done' if ok else 'TERMINATED (error)'}.", flush=True) + finish_wandb() + cleanup_distributed() + + +if __name__ == "__main__": + main() diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_006_mustard_bottle.json b/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_006_mustard_bottle.json new file mode 100644 index 0000000000000000000000000000000000000000..7701b251cfe863680d038ffb6ff4fe8802c2a29c --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_006_mustard_bottle.json @@ -0,0 +1,166 @@ +{ + "obj": "006_mustard_bottle", + "n_views": 1, + "tags": [ + "front" + ], + "n_cloud": 24774, + "n_mesh_verts": 25004, + "chosen": "identity", + "candidates": { + "identity": { + "iou": 0.41451333640169197, + "recall": 0.7634213288124647, + "per_view_iou": [ + 0.4145 + ], + "per_view_recall": [ + 0.7634 + ], + "residual": 0.19906635180219204, + "scale": 1.0, + "T": [ + [ + 1.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 1.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 1.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp": { + "iou": 0.03144425607491725, + "recall": 0.03144425607491725, + "per_view_iou": [ + 0.0314 + ], + "per_view_recall": [ + 0.0314 + ], + "residual": 0.00907034917379286, + "scale": 0.10693419013805648, + "T": [ + [ + 0.07872608872409179, + 0.05367219675487408, + 0.04854296313776607, + 0.02820726183803618 + ], + [ + -0.0707815614000623, + 0.042171131562565994, + 0.06816514687862114, + 0.028689621435054016 + ], + [ + 0.015069656282001508, + -0.08231532043745483, + 0.06657337682064415, + -0.06276081323907015 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp_pre": { + "iou": 0.029991119722289496, + "recall": 0.029991119722289496, + "per_view_iou": [ + 0.03 + ], + "per_view_recall": [ + 0.03 + ], + "residual": 0.008921026959299059, + "scale": 0.10515850318713249, + "T": [ + [ + 0.07450108130600988, + 0.052164540592408444, + 0.05278977535069112, + 0.0275546839060073 + ], + [ + -0.07356981540034464, + 0.04206675700795234, + 0.06225898336265405, + 0.03417775484087342 + ], + [ + 0.009766367735890258, + -0.08104048033035773, + 0.0662975821703208, + -0.06033333657294732 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp_rigid": { + "iou": 0.37622390685076507, + "recall": 0.9755792362961169, + "per_view_iou": [ + 0.3762 + ], + "per_view_recall": [ + 0.9756 + ], + "residual": 0.10572560731815549, + "scale": 1.000000000000003, + "T": [ + [ + 0.8773369447895822, + 0.44799872057701506, + 0.1719797420298292, + 0.021002179060492628 + ], + [ + -0.3313392455198909, + 0.30629436309640756, + 0.8924113779606792, + 0.0819928998013424 + ], + [ + 0.3471227300042422, + -0.8399291098043518, + 0.41716291879513717, + 0.022728955305292987 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + } + }, + "sec": 17.5, + "mesh_out_verts": 25004 +} \ No newline at end of file diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_010_potted_meat_can.json b/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_010_potted_meat_can.json new file mode 100644 index 0000000000000000000000000000000000000000..476e3265752f935492947524d33a299bd315f254 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_010_potted_meat_can.json @@ -0,0 +1,149 @@ +{ + "obj": "010_potted_meat_can", + "n_views": 4, + "tags": [ + "front", + "side", + "back", + "oside" + ], + "n_cloud": 42015, + "n_mesh_verts": 24954, + "chosen": "icp", + "candidates": { + "identity": { + "iou": 0.3584527931135395, + "recall": 0.9911518419994644, + "per_view_iou": [ + 0.3897, + 0.2942, + 0.3254, + 0.4246 + ], + "per_view_recall": [ + 0.9993, + 0.9975, + 0.9967, + 0.9711 + ], + "residual": 0.14478186225553, + "scale": 1.0, + "T": [ + [ + 1.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 1.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 1.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp": { + "iou": 0.5232152509908132, + "recall": 0.643332399423155, + "per_view_iou": [ + 0.474, + 0.5054, + 0.6447, + 0.4688 + ], + "per_view_recall": [ + 0.5525, + 0.6499, + 0.8059, + 0.5651 + ], + "residual": 0.04045626030255339, + "scale": 0.5529621244891169, + "T": [ + [ + 0.4746466365032733, + -0.04742934059326271, + -0.2796929373912288, + 0.023791252090910257 + ], + [ + 0.10615979184625245, + 0.5352634808712737, + 0.08938800680319457, + -0.10405353082678721 + ], + [ + 0.2630737162295109, + -0.1304245943676875, + 0.4685603014903585, + 0.041805765391925154 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp_pre": { + "iou": 0.5225645051218218, + "recall": 0.6428727502358454, + "per_view_iou": [ + 0.4723, + 0.5073, + 0.6421, + 0.4686 + ], + "per_view_recall": [ + 0.5509, + 0.6524, + 0.8026, + 0.5656 + ], + "residual": 0.04035546725976676, + "scale": 0.5517642340756601, + "T": [ + [ + 0.4731595108587962, + -0.05240122420183927, + -0.2789587048132989, + 0.027624233574403967 + ], + [ + 0.11284518908714833, + 0.5323097139298607, + 0.0914117156654218, + -0.1048185029319897 + ], + [ + 0.26044156847333505, + -0.13544094714361263, + 0.4671934387949916, + 0.044804599013305 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + } + }, + "sec": 15.0, + "mesh_out_verts": 24954 +} \ No newline at end of file diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_019_pitcher_base.json b/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_019_pitcher_base.json new file mode 100644 index 0000000000000000000000000000000000000000..24d353a150aa65562e03b98fd2c5894b3fc8b3b6 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_019_pitcher_base.json @@ -0,0 +1,193 @@ +{ + "obj": "019_pitcher_base", + "n_views": 4, + "tags": [ + "front", + "side", + "back", + "oside" + ], + "n_cloud": 80000, + "n_mesh_verts": 24994, + "chosen": "icp", + "candidates": { + "identity": { + "iou": 0.3935959178032537, + "recall": 0.8332771431418656, + "per_view_iou": [ + 0.5319, + 0.3151, + 0.2901, + 0.4372 + ], + "per_view_recall": [ + 0.8549, + 0.8437, + 0.8094, + 0.8251 + ], + "residual": 0.14949471348053134, + "scale": 1.0, + "T": [ + [ + 1.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 1.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 1.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp": { + "iou": 0.712099729387928, + "recall": 0.9256736044371818, + "per_view_iou": [ + 0.7678, + 0.7133, + 0.6167, + 0.7507 + ], + "per_view_recall": [ + 0.9114, + 0.9739, + 0.9292, + 0.8882 + ], + "residual": 0.027950027457136407, + "scale": 0.6988601147111997, + "T": [ + [ + 0.6931088961825547, + -0.08876346327860371, + 0.011250135711994746, + 0.037620594140316044 + ], + [ + 0.028294133981600328, + 0.1340784252547633, + -0.6852940082895699, + -0.003546931807771 + ], + [ + 0.08488203549032765, + 0.6801099053845224, + 0.13656872476166795, + 0.06036142647091183 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp_pre": { + "iou": 0.33659291868687624, + "recall": 0.39401098046669497, + "per_view_iou": [ + 0.2847, + 0.35, + 0.3372, + 0.3745 + ], + "per_view_recall": [ + 0.2911, + 0.4427, + 0.4549, + 0.3873 + ], + "residual": 0.056492519497662104, + "scale": 0.5031946014374609, + "T": [ + [ + 0.5026547302363276, + 0.021704911280682956, + 0.00848091464099161, + 0.07200919052459281 + ], + [ + -0.021097831203238066, + 0.5016421341780986, + -0.03338948414646603, + 0.025162122909830716 + ], + [ + -0.009894978795822662, + 0.03299807508545131, + 0.5020139672868577, + 0.05202800671278107 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp_rigid": { + "iou": 0.37344364146457587, + "recall": 0.9321803512280858, + "per_view_iou": [ + 0.4034, + 0.373, + 0.301, + 0.4163 + ], + "per_view_recall": [ + 0.9412, + 0.9757, + 0.8901, + 0.9217 + ], + "residual": 0.0926331827344291, + "scale": 1.0000000000000016, + "T": [ + [ + 0.9301505615877889, + 0.3656573518204674, + -0.033386132414026694, + 0.04100138523158287 + ], + [ + -0.10912847390599857, + 0.1884858875660527, + -0.9759938761956384, + 0.00018402695270738667 + ], + [ + -0.350586521362235, + 0.9114646297295813, + 0.21522388294761277, + 0.06573648215918057 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + } + }, + "sec": 55.0, + "mesh_out_verts": 24994 +} \ No newline at end of file diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_021_bleach_cleanser.json b/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_021_bleach_cleanser.json new file mode 100644 index 0000000000000000000000000000000000000000..445c9cf5ca442cfd1e3a38cdf9a6e29f67cc7c49 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/pp_021_bleach_cleanser.json @@ -0,0 +1,175 @@ +{ + "obj": "021_bleach_cleanser", + "n_views": 2, + "tags": [ + "front", + "side" + ], + "n_cloud": 45942, + "n_mesh_verts": 25008, + "chosen": "icp_rigid", + "candidates": { + "identity": { + "iou": 0.21261647225942396, + "recall": 0.4009642214018924, + "per_view_iou": [ + 0.2536, + 0.1716 + ], + "per_view_recall": [ + 0.457, + 0.3449 + ], + "residual": 0.194784366520237, + "scale": 1.0, + "T": [ + [ + 1.0, + 0.0, + 0.0, + 0.0 + ], + [ + 0.0, + 1.0, + 0.0, + 0.0 + ], + [ + 0.0, + 0.0, + 1.0, + 0.0 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp": { + "iou": 0.0834616942444708, + "recall": 0.08365429567614885, + "per_view_iou": [ + 0.0796, + 0.0873 + ], + "per_view_recall": [ + 0.0796, + 0.0877 + ], + "residual": 0.010853137810511672, + "scale": 0.19037809456645594, + "T": [ + [ + 0.06535387719470599, + 0.1711294942740083, + 0.05184000208233158, + 0.06063754567634928 + ], + [ + -0.1420780015885707, + 0.016186919543308274, + 0.12568072243208542, + 0.0074291570221916335 + ], + [ + 0.10856579153200956, + -0.08183213742763558, + 0.13326961050369726, + -0.017893747367960015 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp_pre": { + "iou": 0.08507236525386253, + "recall": 0.08536072958608756, + "per_view_iou": [ + 0.073, + 0.0972 + ], + "per_view_recall": [ + 0.0732, + 0.0975 + ], + "residual": 0.011728174574659918, + "scale": 0.19874512044914178, + "T": [ + [ + 0.053510165039480924, + 0.18039652367680933, + 0.06397952317041679, + 0.06818434324253528 + ], + [ + -0.17646458234954784, + 0.020763637707746176, + 0.08904350300659158, + 0.059036085461495864 + ], + [ + 0.0741386290404061, + -0.08078111472867812, + 0.1657633798205895, + 0.012296620365243998 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + }, + "icp_rigid": { + "iou": 0.4082755028454133, + "recall": 0.9737937136423289, + "per_view_iou": [ + 0.414, + 0.4026 + ], + "per_view_recall": [ + 0.9896, + 0.9579 + ], + "residual": 0.08446654825847617, + "scale": 1.0000000000000089, + "T": [ + [ + 0.6267967488439091, + 0.7788226692218856, + 0.023687244348532394, + 0.014211984057836132 + ], + [ + -0.21319682139823912, + 0.1421822906600067, + 0.9666081478905366, + 0.0776061292780723 + ], + [ + 0.7494484311708227, + -0.610916889706608, + 0.25516191503986724, + 0.000257587834327642 + ], + [ + 0.0, + 0.0, + 0.0, + 1.0 + ] + ] + } + }, + "sec": 21.0, + "mesh_out_verts": 25008 +} \ No newline at end of file diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/__pycache__/__init__.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/__pycache__/__init__.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d9f69e2a1b0a6bcb19100c88f781090f7de29da0 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/__pycache__/__init__.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f7cf42bf1e2c2613673b59bfaf3cd07064bd6d56 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..62d6987b4d6a27fe27684f2a83db3bcf9185ec5c Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/__pycache__/utils.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/__pycache__/utils.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7821f2b3051cb0c6d3ed3276309344c1870e736d Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/__pycache__/utils.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..4d0da41320c83f55e95069d30454cc9b4a39e74f Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ac00447420e796ff9dfffcab3fb8cde0d03c2dcc Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/img_and_mask_transforms.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/img_and_mask_transforms.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..726161d33b6bc299f6becda846919cd741d8f83e Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/img_and_mask_transforms.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/img_processing.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/img_processing.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f1d0430c6fe78c8d481fa0c484020a826872ef2a Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/img_processing.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/preprocessor.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/preprocessor.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..289c6df2bfbe0d14a74028b552a06d4918a5e241 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/data/dataset/tdfy/__pycache__/preprocessor.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..05846855ea8fd55f296310528dfaa253dbe0a942 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/__pycache__/io.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/__pycache__/io.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..f0aca9470488ff77be6887f7684e486d3d38e84a Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/__pycache__/io.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..919613fa9ff907e56b3a88e03e9b48f8dcd58b98 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a9087c82a8a1c897a3844cb4dbfd1023087f0205 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c8c77c4f63b3f5fd9c060f32013404d2d4572be5 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/dino.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/dino.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..42497a448082eed24d299052496b23cc8dede666 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/dino.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/embedder_fuser.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/embedder_fuser.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..4be16ef2484dee879a119213c09794a508c8d684 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/embedder_fuser.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/point_remapper.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/point_remapper.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..dea82ab51ebd5116a195593b39fc50f38344e10e Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/point_remapper.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/pointmap.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/pointmap.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..aa5ad3e32b42a9dfe6ab1b983d02dfacf8600e94 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/dit/embedder/__pycache__/pointmap.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2b776206a0fcff5f6d10abf3365e5588d0b4ca99 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/__pycache__/base.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/__pycache__/base.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..60c85581286d8cd2deaee74bec031895cd8950a2 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/__pycache__/base.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/__pycache__/classifier_free_guidance.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/__pycache__/classifier_free_guidance.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d67b90baeb5b0288f40ec5fc4864daf4e37f8f73 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/__pycache__/classifier_free_guidance.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/flow_matching/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/flow_matching/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..4b03cc196f865a4e0647f6960d1def5b72e43c7d Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/flow_matching/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/flow_matching/__pycache__/model.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/flow_matching/__pycache__/model.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..335073279fb4351e0d1a4e2613e2d858b98f619d Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/flow_matching/__pycache__/model.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/flow_matching/__pycache__/solver.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/flow_matching/__pycache__/solver.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c567de516ba01947aad278a387a00df987f57746 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/generator/flow_matching/__pycache__/solver.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a449e3382717abb1cb222ec2656edee116c9ca2a Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..12c41606c62e58ab0ac2073ae2b88ad58c35bad2 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/mm_latent.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/mm_latent.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..88e1d381cc0b13a33d80a04253f49ba4aaf69b5f Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/mm_latent.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/mot_sparse_structure_flow.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/mot_sparse_structure_flow.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6fe465a5deedb579ecaadc5e22518abd1820e2d3 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/mot_sparse_structure_flow.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/sparse_structure_flow.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/sparse_structure_flow.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..53dcc80f13efe33dc6180dc818a50d605a09c241 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/sparse_structure_flow.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/sparse_structure_vae.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/sparse_structure_vae.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..d8ddbe262032ed2752653ac186a7f95ea0dec80f Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/sparse_structure_vae.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/structured_latent_flow.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/structured_latent_flow.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..b3eec004380a014feedc4d9f9830a63cbf29179b Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/structured_latent_flow.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/timestep_embedder.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/timestep_embedder.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..00f4b66851017ca624e37463886291c91b5c42d4 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/__pycache__/timestep_embedder.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/structured_latent_vae/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/structured_latent_vae/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c5c97610bcd52603e3cf01fc82abc927b783c867 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/structured_latent_vae/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/structured_latent_vae/__pycache__/base.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/structured_latent_vae/__pycache__/base.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..24fe2e8ee1e90adae0f36b369fd51e50e9b46da8 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/structured_latent_vae/__pycache__/base.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/structured_latent_vae/__pycache__/encoder.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/structured_latent_vae/__pycache__/encoder.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..75dfea0c1457ad279683257d1a49ec59528161a2 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/models/structured_latent_vae/__pycache__/encoder.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ca7c42d62c100c03a38a74fbd3322e6684f105c6 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/norm.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/norm.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0f2b08f4735095db92de04cb59f5f4a60f2a43b8 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/norm.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/spatial.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/spatial.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..578f89527314ef9643ca8ed4885103284f7149f4 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/spatial.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/utils.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/utils.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..882f24b6edff425bc47289a812492f81732097cd Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/__pycache__/utils.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/attention/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/attention/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2276b94a0e94bcc4fdcfc78b86cd6406d3f4f272 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/attention/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/attention/__pycache__/full_attn.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/attention/__pycache__/full_attn.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..39055cf71d00666c4a282334aca05f3574832545 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/attention/__pycache__/full_attn.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/attention/__pycache__/modules.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/attention/__pycache__/modules.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..07c4bbff074159c2dfc4c6f353c1047167e9e785 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/attention/__pycache__/modules.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..71e80bed377bbed3f139301884aa1c6e1e7784b0 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/basic.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/basic.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6fd36d64df6d670fac5c9de749e0799f75d19d7e Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/basic.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/linear.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/linear.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7a572f5e7f11c1653e1f99d9f7196c4188f20505 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/linear.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/nonlinearity.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/nonlinearity.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..013dceaf42f894669bcb87a11ea90dfd4cfbc2c9 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/nonlinearity.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/norm.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/norm.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6c3ef9daeb32daf33a1f5f6b71199868cc219650 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/norm.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/spatial.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/spatial.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..5a218edd7933d6b52866b70554bec04d11069d72 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/__pycache__/spatial.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..64eb3f82170fc5776453cd0e9ee25f9f9cac35b4 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/full_attn.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/full_attn.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3e3882184f2505a9d2e6444f2fbf13edd66e4af0 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/full_attn.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/masked_sdpa.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/masked_sdpa.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..9392c7628d4aac445904cd1b5c915729aa4d6cbc Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/masked_sdpa.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/modules.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/modules.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..391138cd4cf44e4b5dab85eaadc32f1e4df2235f Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/modules.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/serialized_attn.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/serialized_attn.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7224a1e38def96118eee55618fe095f79605a499 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/serialized_attn.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/windowed_attn.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/windowed_attn.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..eb431bc50c62f35db4907441ed19c08a3782e850 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/attention/__pycache__/windowed_attn.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/conv/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/conv/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..097c545c5c431858d4f14e814383253b84b9929e Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/conv/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/conv/__pycache__/conv_spconv.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/conv/__pycache__/conv_spconv.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..906aea966f8289317c0966f7ecdaeabb621c1d9e Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/conv/__pycache__/conv_spconv.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/transformer/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/transformer/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..bb033c6b74caeab5cb89661440aee598fbe0552b Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/transformer/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/transformer/__pycache__/blocks.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/transformer/__pycache__/blocks.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..11a18dfa92760883562e84f9ebe390e939e6cc87 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/transformer/__pycache__/blocks.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/transformer/__pycache__/modulated.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/transformer/__pycache__/modulated.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..434e04524089324651d03702df80ce502a1bbe3c Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/sparse/transformer/__pycache__/modulated.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/transformer/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/transformer/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e1c748727cf655c2bbe292c6fd0a1a3c20881b6f Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/transformer/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/transformer/__pycache__/blocks.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/transformer/__pycache__/blocks.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..0b6e85db4743f88d7f74e9fc084b978d40a62dd7 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/transformer/__pycache__/blocks.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/transformer/__pycache__/modulated.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/transformer/__pycache__/modulated.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..754821abb877ef9bf59f3ffd5b1d9c64c699ca16 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/backbone/tdfy_dit/modules/transformer/__pycache__/modulated.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/layers/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/layers/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ae4c607a3b1b829d6b3c89a1f81c16391c1697c8 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/layers/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/layers/llama3/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/layers/llama3/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..4ca82b4fec6d34a0b04695b0b24b4bb5044f574a Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/layers/llama3/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/layers/llama3/__pycache__/ff.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/layers/llama3/__pycache__/ff.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..573dc6618c11a34d649129da23d8409405a31572 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/model/layers/llama3/__pycache__/ff.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/pipeline/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/pipeline/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1cd878b81c7e63f58ca0a7c0b93b1f6d7c0807c5 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/pipeline/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/pipeline/__pycache__/inference_utils.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/pipeline/__pycache__/inference_utils.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8cb24561859c659b2ffade2acf38baf22a9961a0 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/pipeline/__pycache__/inference_utils.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/__pycache__/__init__.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/__pycache__/__init__.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..61184ac17ca7b616771f5f94efd052d81f6898d7 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/__pycache__/__init__.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..6a8f4ba74e3ec277bf95a9699c94354349d98d97 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/__init__.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/__init__.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..002cd480d2027cc06e67a56a506eb87227e8d336 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/__init__.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/__init__.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/__init__.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..ef65a1f89b49c1d74f36c981e4a4557027d8b894 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/__init__.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/checkpointing.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/checkpointing.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..7ba3a0217ac8504627bfbef6f84a9a4f5e9c9964 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/checkpointing.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/checkpointing.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/checkpointing.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e76eb449244e238d5386f52c84fbce5d44861522 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/checkpointing.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/distributed.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/distributed.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..49caec844cbd1ca289e311043581c92726515a7b Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/distributed.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/train_utils.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/train_utils.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..54d193e12abc4989c18a97578056466308698ad7 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/sam3d_objects/training/stage1_flow_lora/__pycache__/train_utils.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/encode_latent_sam3d.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/encode_latent_sam3d.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..911738c69491eeeaa7c7e61db1401658cb646928 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/encode_latent_sam3d.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/render.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/render.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..e95664aa170dc24f522d40398d505d8ec53aa643 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/render.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/utils.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/utils.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..120ea091d4f39407eea2def15d93f06ce76f177e Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/utils.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/voxelize.cpython-310.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/voxelize.cpython-310.pyc new file mode 100644 index 0000000000000000000000000000000000000000..03828a7a53c23fa341b1a00175d5736465bc506a Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/__pycache__/voxelize.cpython-310.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/blender_script/__pycache__/render.cpython-39.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/blender_script/__pycache__/render.cpython-39.pyc new file mode 100644 index 0000000000000000000000000000000000000000..a8dc8916375aa1f60fadedac7716694c287dc816 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/slat_official/blender_script/__pycache__/render.cpython-39.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tests/test_tar_cond_variants.py b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tests/test_tar_cond_variants.py index 4b13c633cd2b1d45e20fe53b53e466e6a9ddbb87..ed024203c72d292eb97c7bdd8f86824356e2c718 100644 --- a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tests/test_tar_cond_variants.py +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tests/test_tar_cond_variants.py @@ -5,10 +5,11 @@ Checks 1. knobs OFF -> same index and byte-identical items vs the pre-change module (when the original module path is given). - 2. knobs ON, p_recgen_image=0 -> white-cond objects stay 'white'; latent-only - objects that have recgen// images are INDEXED and come out 'recgen'. - 3. p_recgen_image=1 -> objects with both sources come out 'recgen'; the anchor - item["slat_input"]["image"] differs from the white variant; x1 identical. + 2. knobs ON -> latent-only objects that have recgen// images are INDEXED + and always come out 'recgen'. + 3. objects with BOTH sources: over several epochs the pooled anchor yields both + 'white' and 'recgen' samples; the RecGen item["image"] differs from the + white one; x1 identical. (Full pooled-source evidence: verify_cond_pool.py.) 4. cond_cameras_fallback -> objects without cameras//cond_cameras.json are indexed and their cameras equal fx=W/(2tan(fov/2)). 5. RecGen cameras: projected SLAT voxels land inside the RecGen mask @@ -52,10 +53,8 @@ if orig: check(all(same(old[i], legacy[i]) for i in ids), f"knobs OFF: {len(ids)} sampled items byte-identical to original module") -on0 = build(TarSlatFlowDataset, "on0", use_recgen_images=True, p_recgen_image=0.0, - cond_cameras_fallback=True) -on1 = build(TarSlatFlowDataset, "on1", use_recgen_images=True, p_recgen_image=1.0, - cond_cameras_fallback=True) +on0 = build(TarSlatFlowDataset, "on0", use_recgen_images=True, cond_cameras_fallback=True) +on1 = on0 leg = set(legacy._index) new = [s for s in on0._index if s not in leg] both = [s for s in on0._index if s in leg and on0._members[s].get("rg")] @@ -86,11 +85,16 @@ def save(t, path): Image.fromarray(a[..., 0] if a.shape[-1] == 1 else a).save(path) for sha in both + rgonly + fb[:2]: - i0 = on0._index.index(sha); i1 = on1._index.index(sha) - a = on0[i0]; b = on1[i1] + i0 = on0._index.index(sha) + draws = [] + for ep in range(40): + on0._epoch = ep; draws.append(on0[i0]) + on0._epoch = 0 + a = next((d for d in draws if d["cond_source"] == "white"), draws[0]) + b = next((d for d in draws if d["cond_source"] == "recgen"), draws[0]) if sha in both: check(a["cond_source"] == "white" and b["cond_source"] == "recgen", - f"{sha[:12]}: p=0 -> white, p=1 -> recgen") + f"{sha[:12]}: pooled anchor gives both white and recgen samples over 40 epochs") check(not torch.equal(a["slat_input"]["image"], b["slat_input"]["image"]), f"{sha[:12]}: item['image'] differs white vs recgen") check(torch.equal(a["x1_feats_raw"], b["x1_feats_raw"]), f"{sha[:12]}: same x1 target") diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tests/verify_cond_pool.py b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tests/verify_cond_pool.py new file mode 100644 index 0000000000000000000000000000000000000000..9a422c88d34b196c0230ed410ec5b5c75d60be20 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tests/verify_cond_pool.py @@ -0,0 +1,216 @@ +"""Evidence run for the ONE-POOL condition rule of TarSlatFlowDataset (2026-09-24). + + python tests/verify_cond_pool.py [n_workers] + +Uses REAL tars (slat tars with latents/ cond/ cameras/ + RecGen tars with +recgen//{cameras.json,NN.jpg,NN_mask.png}) and the production data knobs +(max_views 8, seed mixture .15/.35, filter_bad_views on, use_recgen_images on, +cond_cameras_fallback on). For every draw it identifies the source of EVERY +view independently (re-decodes the white PNG and the RecGen jpg+mask of that view +index and checks which one reproduces item["crops"][k] bit-exactly), measures the +per-view voxel-in-mask precision with that source's camera, and records the pool +sizes. Writes draws.jsonl + summary.json + contact sheets to out_dir. +""" +import json, os, sys, collections, multiprocessing as mp, tempfile +import numpy as np, torch +from PIL import Image, ImageDraw + +root, out = sys.argv[1], sys.argv[2] +NW = int(sys.argv[3]) if len(sys.argv) > 3 else 32 +os.makedirs(out, exist_ok=True) +tmp = tempfile.mkdtemp(prefix="pool_") +from mvsam3d.data.slatflow_dataset import (TarSlatFlowDataset, alpha_bbox_crop, + rgba_to_crop) + +KW = dict(tar_roots=[root], split="all", val_fraction=0.0, min_views=1, max_views=8, + seed=0, max_num_voxels=20000, p_seed_none=0.15, p_seed_single=0.35, + filter_bad_views=True, manifest_dir=None) +os.environ["TAR_INDEX_CACHE"] = os.path.join(tmp, "on.pkl") +DS = TarSlatFlowDataset(use_recgen_images=True, cond_cameras_fallback=True, **KW) +os.environ["TAR_INDEX_CACHE"] = os.path.join(tmp, "legacy.pkl") +LEG = TarSlatFlowDataset(**KW) + + +def _same(a, b): + if isinstance(a, torch.Tensor): return isinstance(b, torch.Tensor) and torch.equal(a, b) + if isinstance(a, np.ndarray): return np.array_equal(a, b) + if isinstance(a, dict): return a.keys() == b.keys() and all(_same(a[k], b[k]) for k in a) + if isinstance(a, (list, tuple)): return len(a) == len(b) and all(_same(x, y) for x, y in zip(a, b)) + return a == b + + +def cat(m): + w, r = bool(m.get("cond_pngs")), m.get("rg") is not None + c = "both" if (w and r) else ("white_only" if w else "recgen_only") + return c + ("_fb" if (w and m.get("cond_cam") is None) else "") + + +def cams_for(ds, m, s): + return ds._rg_cameras(m) if s == "recgen" else ds._tar_cameras(ds._get_tar(m["tar"]), m) + + +def precision(coords, v, rgba): + p = (coords.astype(np.float64) + 0.5) / 64 - 0.5 + w2c = np.linalg.inv(v["c2w"]); xc = p @ w2c[:3, :3].T + w2c[:3, 3] + z = xc[:, 2]; zs = np.where(z == 0, 1e-12, z) + u = np.round(v["fx"] * xc[:, 0] / zs + v["cx"]).astype(int) + q = np.round(v["fy"] * xc[:, 1] / zs + v["cy"]).astype(int) + H, W = rgba.shape[:2] + inf = (z > 0) & (u >= 0) & (u < W) & (q >= 0) & (q < H) + return float((rgba[q[inf], u[inf], 3] > 0).sum() / max(len(p), 1)), float((z > 0).mean()) + + +def draw(job): + idx, epoch = job + ds = DS + ds._epoch = epoch + sha = ds._index[idx]; m = ds._members[sha] + try: + it = ds._getitem_impl(idx) + except Exception as e: + return dict(sha=sha, epoch=epoch, err=repr(e)[:200]) + src = it["cond_source"]; n = int(it["n_subset"]); order = it["view_order"][:n] + srcs_present = [s for s in ("white", "recgen") + if (s == "white" and m.get("cond_pngs")) or (s == "recgen" and m.get("rg"))] + cams = {s: cams_for(ds, m, s) for s in srcs_present} + coords = it["coords"][:, 1:].numpy() + views = [] + for k, i in enumerate(order): + got = it["crops"][k].numpy(); match = []; prec = None + for s in srcs_present: + if i >= cams[s][1]: + continue + rgba = ds._load_view_rgba(m, s, i) + pm = rgba_to_crop(alpha_bbox_crop(rgba)[0])["premult"] + if np.array_equal(pm, got): + match.append(s); prec = precision(coords, cams[s][0][i], rgba) + views.append(dict(i=int(i), match=match, prec=prec)) + rgq = ds._valid_cache.get((sha, "rgqc")) + nv = {s: len(ds._valid_cache.get(sha if s == "white" else (sha, s), [])) + for s in srcs_present} + if "recgen" in srcs_present: + # pool size actually used for RecGen = QC-valid (∩ bad-view gate) + fb = ds._valid_cache.get((sha, "recgen"), list(range(cams["recgen"][1]))) + inter = [i for i in fb if i in set(rgq or [])] + nv["recgen"] = len(inter) if inter else len(rgq or []) + if src == "recgen": + bad = [int(i) for i in order if i not in set(rgq or [])] + else: + bad = [] + return dict(sha=sha, epoch=epoch, cat=cat(m), anchor_src=src, n_sub=n, rg_qc_valid=rgq, + recgen_views_failing_qc_used=bad, + seed_mode=float(it["seed_mode"]), views=views, n_valid=nv, + n_views={s: cams[s][1] for s in srcs_present}) + + +def main(): + rng = np.random.default_rng(0) + by = collections.defaultdict(list) + for j, s in enumerate(DS._index): + by[cat(DS._members[s])].append(j) + print({k: len(v) for k, v in by.items()}, flush=True) + both = list(rng.permutation(by["both"])) + jobs = [(int(j), e) for j in both[:400] for e in range(5)] # 2000 + jobs += [(int(j), e) for j in both[400:420] for e in range(100, 160)] # 1200 + jobs += [(int(j), e) for j in rng.permutation(by["recgen_only"])[:150] for e in range(2)] + jobs += [(int(j), e) for j in rng.permutation(by["white_only"])[:100] for e in range(2)] + for k in ("white_only_fb", "both_fb"): + jobs += [(int(j), e) for j in rng.permutation(by[k])[:50] for e in range(2)] + with mp.get_context("fork").Pool(NW) as pool: + recs = pool.map(draw, jobs, chunksize=4) + with open(os.path.join(out, "draws.jsonl"), "w") as f: + for r in recs: + f.write(json.dumps(r) + "\n") + + S = collections.Counter(); P = collections.defaultdict(list) + per_obj = collections.defaultdict(lambda: [0, 0, None]) + mixed = [] + for r in recs: + if "err" in r: + S["errors"] += 1; continue + S["draws"] += 1; S["draws_" + r["cat"]] += 1 + ids = [tuple(v["match"]) for v in r["views"]] + if any(len(x) != 1 for x in ids): + S["views_unidentified_or_ambiguous"] += 1 + vs = set(x[0] for x in ids if len(x) == 1) + if vs != {r["anchor_src"]}: + mixed.append(r) + S[f"anchor_{r['anchor_src']}__{r['cat']}"] += 1 + S["recgen_views_used_that_fail_qc"] += len(r["recgen_views_failing_qc_used"]) + if r.get("rg_qc_valid") == []: + S["draws_obj_recgen_set_dropped_by_qc"] += 1 + S["aux_views_" + r["anchor_src"]] += max(0, len(ids) - 1) + for k, v in enumerate(r["views"]): + if v["prec"] is not None: + P[(r["anchor_src"], "anchor" if k == 0 else "aux")].append(v["prec"][0]) + if r["cat"].startswith("both"): + o = per_obj[r["sha"]]; o[0] += 1; o[1] += r["anchor_src"] == "recgen" + nw, nr = r["n_valid"].get("white", 0), r["n_valid"].get("recgen", 0) + if r.get("rg_qc_valid") == []: + nr = 0 + o[2] = nr / max(1, nw + nr) + exp = sum(o[2] * o[0] for o in per_obj.values()); obs = sum(o[1] for o in per_obj.values()) + ntot = sum(o[0] for o in per_obj.values()) + var = sum(o[0] * o[2] * (1 - o[2]) for o in per_obj.values()) + heavy = [(s[:16], o[0], o[1] / o[0], o[2]) for s, o in per_obj.items() if o[0] >= 50] + summ = dict( + counts=dict(S), mixed_source_samples=len(mixed), + precision={f"{a}/{b}": dict(n=len(v), mean=float(np.mean(v)), p05=float(np.percentile(v, 5)), + median=float(np.median(v)), min=float(np.min(v))) + for (a, b), v in P.items()}, + pool_ratio_check=dict(draws_both=ntot, observed_recgen_anchors=obs, + expected_recgen_anchors=exp, z=(obs - exp) / max(np.sqrt(var), 1e-9)), + per_object_60_draws=[dict(obj=h[0], draws=h[1], recgen_anchor_frac=h[2], + pool_ratio_n_r_over_n=h[3]) for h in heavy]) + json.dump(summ, open(os.path.join(out, "summary.json"), "w"), indent=1) + print(json.dumps(summ, indent=1)[:6000], flush=True) + + # ---- white-only byte-identity vs legacy on the SAME index list ------------- # + import copy + same = diff = 0 + ON = copy.copy(DS); ON._index = LEG._index; ON._valid_cache = {} + LEG._valid_cache = {} + for j in list(rng.permutation(len(LEG._index)))[:60]: + for e in (0, 3): + LEG._epoch = e; ON._epoch = e + a = LEG._getitem_impl(int(j)); b = ON._getitem_impl(int(j)) + if cat(ON._members[LEG._index[j]]) != "white_only": + continue + b.pop("cond_source") + ok = _same(a, b) + same += ok; diff += (not ok) + print(f"white-only byte-identity vs legacy (same index list): identical={same} differ={diff}", flush=True) + summ["white_only_identity_vs_legacy"] = dict(identical=same, differ=diff) + json.dump(summ, open(os.path.join(out, "summary.json"), "w"), indent=1) + + # ---- contact sheets ------------------------------------------------------ # + picks = [] + for want in ("white", "recgen"): + picks += [r for r in recs if "err" not in r and r["cat"] == "both" + and r["anchor_src"] == want and r["n_sub"] >= 3][:3] + picks += [r for r in recs if "err" not in r and r["cat"] == "recgen_only" and r["n_sub"] >= 3][:1] + for r in picks: + j = DS._index.index(r["sha"]); DS._epoch = r["epoch"]; it = DS._getitem_impl(j) + m = DS._members[r["sha"]]; s = it["cond_source"]; va = cams_for(DS, m, s)[0] + c = it["coords"][:, 1:].numpy() + tiles = [] + a = it["slat_input"]["image"][0].permute(1, 2, 0).numpy() + tiles.append(("anchor item['image']", Image.fromarray((np.clip(a, 0, 1) * 255).astype(np.uint8)).resize((256, 256)))) + for k, i in enumerate(it["view_order"][: it["n_subset"]]): + rgba = DS._load_view_rgba(m, s, i); ov = rgba[..., :3].copy() + v = va[i]; p = (c + 0.5) / 64 - 0.5; w2c = np.linalg.inv(v["c2w"]); xc = p @ w2c[:3, :3].T + w2c[:3, 3] + u = np.round(v["fx"] * xc[:, 0] / xc[:, 2] + v["cx"]).astype(int); q = np.round(v["fy"] * xc[:, 1] / xc[:, 2] + v["cy"]).astype(int) + H, W = ov.shape[:2]; kk = (u >= 0) & (u < W) & (q >= 0) & (q < H) + ov[q[kk], u[kk]] = (0.5 * ov[q[kk], u[kk]] + [127, 0, 127]).astype(np.uint8) + y0, y1, x0, x1 = [int(b) for b in alpha_bbox_crop(rgba)[1]] + pad = 40; crop = ov[max(0, y0 - pad):y1 + pad, max(0, x0 - pad):x1 + pad] + tiles.append((f"{'anchor' if k == 0 else 'aux'} {s} v{i}", Image.fromarray(crop).resize((256, 256)))) + sheet = Image.new("RGB", (258 * len(tiles), 280), "white"); d = ImageDraw.Draw(sheet) + for t, (lab, im) in enumerate(tiles): + sheet.paste(im, (258 * t, 22)); d.text((258 * t + 2, 4), lab, fill="black") + sheet.save(os.path.join(out, f"contact_{r['cat']}_{s}_{r['sha'][:16]}_ep{r['epoch']}.png")) + print("DONE", flush=True) + + +if __name__ == "__main__": + main() diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/build_canonical_latents.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/build_canonical_latents.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..50fa97d9039fda9841f29378da188d30630370cb Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/build_canonical_latents.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/build_recon_trained.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/build_recon_trained.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..bd40031afb8fb65c115bb9eadff3575db03a5955 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/build_recon_trained.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/gen_x1_official.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/gen_x1_official.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..73d087930896b2f94a0fc18cc755f11f9af5ef32 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/gen_x1_official.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/merge_extra_cond_views.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/merge_extra_cond_views.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..8623951a2405758825dff732cd82619ee6df899d Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/merge_extra_cond_views.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/preprocess_ssflow_toys4k.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/preprocess_ssflow_toys4k.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..17442d1687a7f22d7a2524378ab73f550651cea3 Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/preprocess_ssflow_toys4k.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/smoke_slatflow_forward.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/smoke_slatflow_forward.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c638e6ca566b2530b363d37b103a0963ea9004cb Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/smoke_slatflow_forward.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/verify_batched_val.cpython-311.pyc b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/verify_batched_val.cpython-311.pyc new file mode 100644 index 0000000000000000000000000000000000000000..2e59761c7c355a973bd11e2298c33c2ef3a00e6c Binary files /dev/null and b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/__pycache__/verify_batched_val.cpython-311.pyc differ diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/_dbg_ours_tokens.py b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/_dbg_ours_tokens.py new file mode 100644 index 0000000000000000000000000000000000000000..5ac2b19dffc78fac383f450039066008666b2331 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/_dbg_ours_tokens.py @@ -0,0 +1,44 @@ +"""Train-env side of the appearance-seed parity debug: compare our DINO raw +tokens / uv / sampled feats / z_norm against the reference dump.""" +import os, sys +os.environ.setdefault("SPARSE_ATTN_BACKEND", "sdpa") +os.environ.setdefault("SPCONV_ALGO", "native") +import numpy as np, torch, torch.nn.functional as F +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) +from mvsam3d.train.train_slat_flow import load_config +from mvsam3d.train.val_appforce import _load_model, build_item, load_view, project_visible, crop_uv + +exp, obj, refnpz, dbgnpz, cfgp = sys.argv[1:6] +cfg = load_config(cfgp, []) +model = _load_model(cfg, None, "cuda") +z = np.load(refnpz); d = np.load(dbgnpz) +coords = torch.from_numpy(z["coords"].astype(np.int32)) +item = build_item(exp, obj, ["front"], coords) +dev = "cuda" +raw, normed = model.mv_embedder.dino_prenorm(item["crops"].to(dev)) +n_skip = model.mv_embedder.dino_image.backbone.num_register_tokens + 1 +ours_tok = raw[0, n_skip:, :].permute(1, 0).reshape(1, 1024, 37, 37).float() +ref_tok = torch.from_numpy(d["tokens"]).to(dev) +print("tokens max|d|", float((ours_tok - ref_tok).abs().max()), "scale", float(ref_tok.abs().max()), + "bit", torch.equal(ours_tok, ref_tok)) +# uv +v = load_view(exp, obj, "front") +u, vv, _, vis = project_visible((coords[:, 1:].numpy().astype(np.float64) + 0.5) / 64 - 0.5, v) +print("u max|d|", np.abs(u - d["u"]).max(), "vv", np.abs(vv - d["vv"]).max(), "vis eq", np.array_equal(vis, d["vis"])) +uv_ours = item["uv"][0].to(dev).view(1, -1, 1, 2) +f_ours_reftok = F.grid_sample(ref_tok, uv_ours, mode="bilinear", align_corners=False).squeeze(-1).squeeze(0).permute(1, 0) +f_ours_ourtok = F.grid_sample(ours_tok, uv_ours, mode="bilinear", align_corners=False).squeeze(-1).squeeze(0).permute(1, 0) +ref_f = torch.from_numpy(d["feats"]).to(dev) +print("feats(ref tok, our uv) max|d|", float((f_ours_reftok - ref_f).abs().max()), + "| feats(our tok, our uv)", float((f_ours_ourtok - ref_f).abs().max()), "scale", float(ref_f.abs().max())) +# z_norm +z_vis, cnt = model.encode_visible_latent(coords.to(dev), raw, item["uv"].to(dev), item["vis"].to(dev), + view_idx=torch.arange(1, device=dev)) +zn = z_vis.cpu().numpy() +print("z_norm max|d|", np.abs(zn - z["z_norm"]).max(), "bit", np.array_equal(zn, z["z_norm"])) +print("slat_mean ours", model.slat_mean.cpu().numpy(), "ref", d["slat_mean"]) +print("slat_std ours", model.slat_std.cpu().numpy(), "ref", d["slat_std"]) +# encoder weights fingerprint +enc = model._get_slat_encoder(dev) +sd = enc.state_dict(); k = sorted(sd)[0] +print("enc first key", k, float(sd[k].float().sum()), "n", len(sd)) diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/_dbg_ref_tokens.py b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/_dbg_ref_tokens.py new file mode 100644 index 0000000000000000000000000000000000000000..b01501b3a3c3482e7f278a912d61f7c72f2de126 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/_dbg_ref_tokens.py @@ -0,0 +1,22 @@ +"""Debug dump of the REFERENCE appearance-seed intermediates for one object/1v.""" +import os, sys +ENV_PY="/lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/bin/python" +if not os.environ.get("_VAL_STAGE1_INENV"): + env=dict(os.environ); env["_VAL_STAGE1_INENV"]="1"; env["_APPFORCE_SAM3D_INENV"]="1"; env.pop("PYTHONPATH",None) + os.execve(ENV_PY,[ENV_PY,os.path.abspath(__file__)]+sys.argv[1:],env) +os.environ["_APPFORCE_SAM3D_INENV"]="1" +sys.path.insert(0,"/lp-dev/jonghoon/mv-mesh/metrics") +import batch_appforce_sam3d as baf +import numpy as np, torch +exp, obj, refnpz, out = sys.argv[1:5] +M = baf.Models() +z = np.load(refnpz); coords = torch.from_numpy(z["coords"]).cuda() +png = os.path.join(exp,"inputs",f"{obj}_front.png") +pt = baf.dino_features(M.dino, M.dino_norm, png) # (1,1024,37,37) +v = baf.load_view(exp,obj,"front") +idx = coords[:,1:].cpu().numpy().astype(np.int64); centers=(idx.astype(np.float64)+0.5)/64-0.5 +u,vv,zz,vis = baf.project_visible(centers, v) +f = baf.sample_feats(pt,u,vv,v) # (N,1024) +np.savez(out, tokens=pt.float().cpu().numpy(), u=u, vv=vv, vis=vis, feats=f.float().cpu().numpy(), + slat_mean=M.slat_mean, slat_std=M.slat_std) +print("dumped", out, pt.shape, f.shape, vis.sum()) diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/build_recon_trained_occ.py b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/build_recon_trained_occ.py new file mode 100644 index 0000000000000000000000000000000000000000..c648782a3cdb571b6324061b99184b383e8d2d89 --- /dev/null +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/build_recon_trained_occ.py @@ -0,0 +1,375 @@ +#!/usr/bin/env python +"""Reconstruct HO3D / DexYCB multi-view anchor objects with the TRAINED SLAT flow +and stage them as another Any6D "baseline" under recon_mv/. + +Pipeline per object (all V anchor views): + + 1. EXP LAYOUT -- built by the *existing, validated* adapter code + (any6d_mesh_eval/code[/dexycb]/build_recon_ours.build_exp_mv), imported and + called verbatim: renders/_.npz (object-only depth_mm in canonical + units + real K + shared canonical c2w from mv_cam.compute_canonical), + inputs/_.png (RGBA object crop), npz_v//da3_output.npz + (per-view camera-canonical pointmaps) and hull_occ.npz (occ=ref: the + REFERENCE-view depth shell, the HO3D-validated stage-1 seed). + The ONLY change: the adapter's view cap (_gen_nviews / _VIEW_TAGS, 1|2|4) + is lifted to ALL V views (tags = val_stage1_ref.VIEW_TAGS[:V]); for V<=2 + the tag lists and npz dir names are the previous ones exactly. + + 2. STAGE 1 (frozen, reference env) -- tools/val_stage1_ref.py --mode stage1 + --views V: SAM3D geometry forcing (z1 = ss_enc(occ) injected as the SS + noise) through pipe.sample_sparse_structure_multi_view(multidiffusion) over + the V per-view conditions, + the post-stage-1 base noise draw + the + reference appearance seed. AF_OCC=hull is exported when the adapter wrote + hull_occ.npz, exactly as build_recon_ours.run_generator does. + + 3. STAGE 2 (TRAINED) -- mvsam3d.train.val_appforce.build_item / stage2 with + tags = all V views (anchor = view0 = 'front', non-anchor = the rest, with + anchor-relative Plucker + local soft-mask bias), for both forcing + schedules (seed, reinject) and the EMA weights of --ckpt. + + 4. DECODE (reference env) -- val_stage1_ref.py --mode decode -> vertex-colored + GLB (baf.decode_mesh_arrays + finalize_glb, unchanged). + + 5. STAGE -- recon_mv/ours_trained_/anchors//{mesh_.obj, + color.png, depth.png, mask.png, K.txt, _gt_pose.txt}: the exact + contract build_anchor.py + run_query_ours.py / run_query_dexycb.py consume + (same layout as recon_mv/ours_sam3d). + +Usage (mv-sam3d env, repo on PYTHONPATH): + MIGRATOR_CHECKPOINTS=/migrator/checkpoints SPARSE_ATTN_BACKEND=sdpa \ + SPCONV_ALGO=native PYTHONPATH= CUDA_VISIBLE_DEVICES=2 \ + python tools/build_recon_trained.py --dataset ho3d --gpu 2 \ + --objs 019_pitcher_base,010_potted_meat_can \ + --ckpt /data/mv_mesh_data/ckpt/slatflow_prod_20260902/step_0012500.pt +""" +from __future__ import annotations + +import argparse +import importlib.util +import json +import os +import shutil +import subprocess +import sys +import time +from pathlib import Path + +import numpy as np + +_HERE = os.path.dirname(os.path.abspath(__file__)) +REPO = os.path.dirname(_HERE) +sys.path.insert(0, REPO) +os.environ.setdefault("SPARSE_ATTN_BACKEND", "sdpa") +os.environ.setdefault("SPCONV_ALGO", "native") + +WS = "/lp-dev/jonghoon/any6d_mesh_eval" +CODE = f"{WS}/code" +DCODE = f"{CODE}/dexycb" +REF_PY = "/lp-dev/jonghoon/mv-mesh/envs/mv-sam3d/bin/python" +STAGE1_REF = os.path.join(_HERE, "val_stage1_ref.py") +VIEW_TAGS = ["front", "side", "back", "oside", "top", "bottom"] +ANCHOR_FILES = ["color.png", "depth.png", "mask.png", "K.txt"] + + +def log(msg: str) -> None: + print(f"[trained] {msg}", flush=True) + + +# --------------------------------------------------------------------------- # +# the existing adapter, with its view cap lifted +# --------------------------------------------------------------------------- # +# selection recorded per object so the report can state which views were used +VIEW_SELECTION = {} + + +def load_adapter(dataset: str, max_views: int = 4): + """Import build_recon_ours (HO3D or DexYCB variant), lift its 1|2|4 view cap + to N views, and make the view SUBSET the MOST-VISIBLE ones. + + The model is trained with at most 4 condition views, so `max_views` caps the + subset at 4 by default. When an anchor carries more views than that we keep + view0 (the reference / Any6D anchor frame -- never dropped, it defines the + canonical frame) plus the `max_views-1` remaining views with the largest + `visible_px` in views.json, in descending visibility order. With <= + max_views available this is a strict no-op (original file order).""" + d = DCODE if dataset == "dexycb" else CODE + sys.path.insert(0, CODE) # mv_cam + spec = importlib.util.spec_from_file_location( + f"build_recon_ours_{dataset}", os.path.join(d, "build_recon_ours.py")) + mod = importlib.util.module_from_spec(spec) + sys.modules[spec.name] = mod + spec.loader.exec_module(mod) + mod._VIEW_TAGS = {i: VIEW_TAGS[:i] for i in range(1, len(VIEW_TAGS) + 1)} + mod._gen_nviews = lambda v_avail: int(min(v_avail, max_views, len(VIEW_TAGS))) + + import mv_cam as _mv + if not getattr(_mv, "_ORIG_LOAD_VIEWS", None): + _mv._ORIG_LOAD_VIEWS = _mv.load_views + + def load_views_most_visible(obj, anchor_dir=None): + info, views = _mv._ORIG_LOAD_VIEWS(obj, anchor_dir=anchor_dir) + if len(views) <= max_views: + VIEW_SELECTION[obj] = [v["idx"] for v in views] + return info, views + vis = {rec["view"]: int(rec.get("visible_px", 0)) + for rec in info["views"]} + rest = sorted((v for v in views if v["idx"] != views[0]["idx"]), + key=lambda v: -vis.get(v["idx"], 0)) + sel = [views[0]] + rest[: max_views - 1] + VIEW_SELECTION[obj] = [(v["idx"], vis.get(v["idx"], 0)) for v in sel] + return info, sel + + _mv.load_views = load_views_most_visible + mod.mv_cam = _mv + return mod + + +def anchors_mv_dir(dataset: str) -> str: + return f"{WS}/dexycb/anchors_mv_occ" if dataset == "dexycb" else f"{WS}/anchors_mv" + + +def recon_root(dataset: str) -> str: + return f"{WS}/dexycb/recon_mv_occ" if dataset == "dexycb" else f"{WS}/recon_mv" + + +def all_objects(dataset: str): + if dataset == "dexycb": + return list(json.load(open(f"{WS}/dexycb/manifest_dexycb.json")).keys()) + m = json.load(open(f"{WS}/anchors_mv/manifest_mv.json")) + return list(m.keys()) + + +# --------------------------------------------------------------------------- # +# step 1: exp layout +# --------------------------------------------------------------------------- # +def build_exp(adapter, dataset: str, obj: str, work: str, target: float, + occ: str, max_views: int): + exp = os.path.join(work, "exp") + os.makedirs(exp, exist_ok=True) + ng, info = adapter.build_exp_mv(obj, exp, target, True, occ_mode=occ) + json.dump({"selections": [{"object": obj}]}, + open(os.path.join(exp, "selection.json"), "w")) + Path(os.path.join(exp, "exp_info.json")).write_text( + json.dumps(dict(obj=obj, dataset=dataset, views=ng, + view_selection=VIEW_SELECTION.get(obj), + info={k: v for k, v in info.items()}), indent=1, + default=str)) + log(f"exp {obj}: V={ng} tags={info['tags']} views(idx,visible_px)=" + f"{VIEW_SELECTION.get(obj)} s={info['s']:.4f} " + f"extent={info['extent_m']:.3f}m occ={info['occ_mode']} " + f"hull={info['hull_saved']}") + return exp, ng + + +# --------------------------------------------------------------------------- # +# step 2 / 4: reference-env subprocesses +# --------------------------------------------------------------------------- # +def _ref_env(gpu: str, exp: str, hull: bool): + env = dict(os.environ) + env.pop("PYTHONPATH", None) + env["CUDA_VISIBLE_DEVICES"] = str(gpu) + env["EGL_DEVICE_ID"] = str(gpu) + env["PYOPENGL_PLATFORM"] = "egl" + env["PYTORCH_CUDA_ALLOC_CONF"] = "expandable_segments:True" + for k in ("OMP_NUM_THREADS", "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS"): + env[k] = "8" + if hull: + env["AF_OCC"] = "hull" # same hook build_recon_ours.run_generator uses + else: + env.pop("AF_OCC", None) + return env + + +def run_stage1(obj: str, exp: str, views: int, out_dir: str, gpu: str, seed: int, + logf: str) -> bool: + dump = os.path.join(out_dir, f"{obj}.npz") + if os.path.isfile(dump): + log(f"stage1 {obj}: cached") + return True + hull = os.path.isfile(os.path.join(exp, "hull_occ.npz")) + cmd = [REF_PY, "-u", STAGE1_REF, "--mode", "stage1", "--exp", exp, + "--views", str(views), "--out", out_dir, "--objects", obj, + "--seed", str(seed)] + t0 = time.time() + with open(logf, "ab") as lf: + lf.write((" ".join(cmd) + "\n").encode()) + rc = subprocess.run(cmd, stdout=lf, stderr=subprocess.STDOUT, + env=_ref_env(gpu, exp, hull), cwd="/lp-dev/jonghoon/mv-mesh").returncode + ok = os.path.isfile(dump) + log(f"stage1 {obj}: rc={rc} ok={ok} ({time.time() - t0:.0f}s)") + return ok + + +def run_decode(objs, jobs, gpu: str, logf: str, work_root: str) -> None: + """ONE reference-env decode process for every (object, schedule) pair -- the + models are loaded once. jobs: [(slat_dir, glb_dir), ...].""" + todo = [o for o in objs + if any(os.path.isfile(os.path.join(sd, f"{o}.npz")) + and not os.path.isfile(os.path.join(gd, f"{o}.glb")) + for sd, gd in jobs)] + if not todo or not jobs: + return + # per-process file: two GPU shards share work_root, so a single fixed name + # races (the loser decodes the OTHER shard's object list -> 0 GLBs). + of = os.path.join(work_root, f"_decode_objs_{os.getpid()}.txt") + Path(of).write_text("\n".join(todo) + "\n") + cmd = [REF_PY, "-u", STAGE1_REF, "--mode", "decode", "--objects-file", of, + "--jobs", *[f"{sd}:{gd}" for sd, gd in jobs]] + t0 = time.time() + with open(logf, "ab") as lf: + lf.write((" ".join(cmd) + "\n").encode()) + rc = subprocess.run(cmd, stdout=lf, stderr=subprocess.STDOUT, + env=_ref_env(gpu, "", False), + cwd="/lp-dev/jonghoon/mv-mesh").returncode + log(f"decode {len(todo)} objs x {len(jobs)} jobs: rc={rc} " + f"({time.time() - t0:.0f}s)") + + +# --------------------------------------------------------------------------- # +# step 3: trained stage-2 +# --------------------------------------------------------------------------- # +def load_trained(config: str, ckpt: str, no_ema: bool): + import torch # noqa: F401 + from mvsam3d.train.train_slat_flow import load_config + from mvsam3d.train import val_appforce as V + cfg = load_config(config, []) + model = V._load_model(cfg, ckpt, "cuda", use_ema=not no_ema) + return model, V + + +def sample_object(V, model, exp: str, obj: str, tags, stage1_npz: str, + schedules, out_root: str, seed: int) -> dict: + import torch + z = np.load(stage1_npz) + coords = torch.from_numpy(z["coords"].astype(np.int32)) + base = torch.from_numpy(z["base"].astype(np.float32))[:1] + item = V.build_item(exp, obj, tags, coords) + stats = {} + for s in schedules: + p = os.path.join(out_root, s, "slat", f"{obj}.npz") + if os.path.isfile(p): + stats[s] = "cached" + continue + t0 = time.time() + r = V.stage2(model, item, base, s, generator_seed=seed) + V._write_slat(p, r) + stats[s] = dict(N=int(r["coords"].shape[0]), n_vis=int(r["n_vis"]), + n_pinned=int(r["n_pinned"]), sec=round(time.time() - t0, 1)) + torch.cuda.empty_cache() + return stats + + +# --------------------------------------------------------------------------- # +# step 5: stage into the Any6D baseline contract +# --------------------------------------------------------------------------- # +def stage_mesh(adapter, dataset: str, obj: str, glb: str, method: str) -> str: + import trimesh + dest = f"{recon_root(dataset)}/{method}/anchors/{obj}" + os.makedirs(dest, exist_ok=True) + m = trimesh.load(glb, force="mesh") + m = adapter.simplify_dense_mesh(m) + out = f"{dest}/mesh_{obj}.obj" + m.export(out) + src = f"{anchors_mv_dir(dataset)}/{obj}" + for f in ANCHOR_FILES + [f"{obj}_gt_pose.txt"]: + shutil.copy(f"{src}/{f}", f"{dest}/{f}") + log(f"staged {method}/{obj}: verts={len(m.vertices)} faces={len(m.faces)} -> {out}") + return out + + +# --------------------------------------------------------------------------- # +def main(): + ap = argparse.ArgumentParser(description=__doc__, + formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--dataset", choices=["ho3d", "dexycb"], default="ho3d") + ap.add_argument("--objs", default="", help="comma list (default: all)") + ap.add_argument("--gpu", default=os.environ.get("CUDA_VISIBLE_DEVICES", "0"), + help="PHYSICAL gpu id for the reference-env subprocesses") + ap.add_argument("--ckpt", required=True) + ap.add_argument("--config", default=os.path.join(REPO, "configs", "train_slatflow_prod.yaml")) + ap.add_argument("--no-ema", action="store_true") + ap.add_argument("--schedules", default="seed,reinject") + ap.add_argument("--method-prefix", default="ours_trained") + ap.add_argument("--target", type=float, default=0.7) + ap.add_argument("--occ", choices=["depth", "hull", "ref"], default="ref") + ap.add_argument("--max-views", type=int, default=4, + help="condition-view cap (4 = the training maximum); the " + "subset is view0 + the most-visible others") + ap.add_argument("--seed", type=int, default=42) + ap.add_argument("--work-root", default="") + ap.add_argument("--stage1-only", action="store_true") + ap.add_argument("--skip-stage1", action="store_true") + args = ap.parse_args() + + objs = [o for o in args.objs.split(",") if o] or all_objects(args.dataset) + scheds = [s for s in args.schedules.split(",") if s] + adapter = load_adapter(args.dataset, max_views=args.max_views or 6) + tag = Path(args.ckpt).stem + work_root = args.work_root or (f"{recon_root(args.dataset)}/" + f"{args.method_prefix}/_work/{tag}") + os.makedirs(work_root, exist_ok=True) + os.makedirs(f"{WS}/logs", exist_ok=True) + logf = f"{WS}/logs/trained_{args.dataset}_gpu{args.gpu}.log" + log(f"dataset={args.dataset} objs={objs} ckpt={args.ckpt} scheds={scheds} " + f"gpu={args.gpu} work={work_root}") + + # ---- 1 + 2 : exp layout + frozen stage-1 (reference env) ---------------- + meta = {} + for obj in objs: + work = os.path.join(work_root, obj) + exp, V_n = build_exp(adapter, args.dataset, obj, work, args.target, + args.occ, args.max_views) + s1_dir = os.path.join(work, f"stage1_{V_n}v") + os.makedirs(s1_dir, exist_ok=True) + ok = True + if not args.skip_stage1: + ok = run_stage1(obj, exp, V_n, s1_dir, args.gpu, args.seed, logf) + meta[obj] = dict(exp=exp, views=V_n, stage1=os.path.join(s1_dir, f"{obj}.npz"), + ok=ok and os.path.isfile(os.path.join(s1_dir, f"{obj}.npz")), + work=work) + Path(os.path.join(work_root, "meta.json")).write_text(json.dumps(meta, indent=1)) + if args.stage1_only: + log("stage1-only: done") + return + + # ---- 3 : trained stage-2 ------------------------------------------------ + ready = [o for o in objs if meta[o]["ok"]] + if not ready: + log("no object has a stage-1 dump; abort") + sys.exit(1) + model, V = load_trained(args.config, args.ckpt, args.no_ema) + s2 = {} + for obj in ready: + m = meta[obj] + tags = VIEW_TAGS[: m["views"]] + try: + s2[obj] = sample_object(V, model, m["exp"], obj, tags, m["stage1"], + scheds, m["work"], args.seed) + log(f"stage2 {obj}: {s2[obj]}") + except Exception as e: + import traceback + traceback.print_exc() + log(f"stage2 FAIL {obj}: {e}") + del model + import torch + torch.cuda.empty_cache() + Path(os.path.join(work_root, "stage2.json")).write_text(json.dumps(s2, indent=1)) + + # ---- 4 + 5 : decode + stage -------------------------------------------- + jobs = [(os.path.join(meta[o]["work"], s, "slat"), + os.path.join(meta[o]["work"], s, "glb")) + for s in scheds for o in ready] + run_decode(ready, jobs, args.gpu, logf, work_root) + for s in scheds: + for obj in ready: + glb = os.path.join(meta[obj]["work"], s, "glb", f"{obj}.glb") + if os.path.isfile(glb): + stage_mesh(adapter, args.dataset, obj, glb, + f"{args.method_prefix}_{s}") + else: + log(f"DECODE-FAIL {s}/{obj}") + log("DONE") + + +if __name__ == "__main__": + main() diff --git a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/pack_recgen_images.py b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/pack_recgen_images.py index 5d1ef8a75dff2168ebfbc6215b3a34e83a9828b6..60b313a98f10acf3e4c9e8e80271d2e4da1302be 100644 --- a/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/pack_recgen_images.py +++ b/migrator/code/mv-sam3d-for-6d-v2-ssflow/tools/pack_recgen_images.py @@ -36,8 +36,12 @@ def recgen_cameras(od): c2w,_=rigidify(np.linalg.inv(analytic_w2c_scaled(m))); cams[i]=c2w out={} for i,c2w in cams.items(): - m=meta[i]; out[i]=dict(K=np.asarray(m['intrinsics'],float).tolist(),c2w=c2w.tolist(),width=int(m.get('width',640)),height=int(m.get('height',480))) + m=meta[i]; out[i]=dict(K=np.asarray(m['intrinsics'],float).tolist(),c2w=c2w.tolist(),width=int(m.get('width',640)),height=int(m.get('height',480)), + visible_fraction=float(m.get('visible_fraction',-1.0))) return out,src +def pose_seedfrac(od): + pc=os.path.join(od,'views20/pose_cube.npz') + return float(np.load(pc)['pose_seedfrac']) if os.path.isfile(pc) else None S=os.environ.get('RECGEN_STORE','/lp-dev/jonghoon/ss_flow_data/encoded_store/recgen/objects') @@ -54,8 +58,8 @@ def pack(pairs,out): L=set(os.listdir(cs)) views=sorted(v for v in cams if f'cond_image_{v:02d}.jpg' in L and f'cond_mask_{v:02d}.png' in L) if not views: rows.append((sid,rid,0,src,'noviews')); continue - add_bytes(tf,f'recgen/{sid}/cameras.json',json.dumps(dict(recgen_id=rid,pose_source=src, - frame='OpenCV c2w, camera->SLAT cube [-0.5,0.5]^3, rigid (cube units); K in pixels of the raw WxH RecGen frame', + add_bytes(tf,f'recgen/{sid}/cameras.json',json.dumps(dict(recgen_id=rid,pose_source=src,pose_seedfrac=pose_seedfrac(od), + frame='OpenCV c2w, camera->SLAT cube [-0.5,0.5]^3, rigid (cube units); K in pixels of the raw WxH RecGen frame; visible_fraction = RecGen view_metadata (modal/amodal)', views={str(v):cams[v] for v in views})).encode()) for v in views: add_bytes(tf,f'recgen/{sid}/{v:02d}.jpg',open(f'{cs}/cond_image_{v:02d}.jpg','rb').read())