{ "model_id": "2026_W41_unit", "task": "unit-encoding", "model_class": "UnitAutoencoder", "vintage": "2026_W41", "training": { "random_seed": 0, "training_size": 572, "training_units": 56497, "validation_size": 75, "validation_units": 7671, "testing_size": 149, "testing_units": 15880 }, "environment": { "python": "3.11.9", "xgboost": "3.2.0", "scikit_learn": "1.8.0", "numpy": "2.3.5", "pandas": "3.0.5", "torch": "2.7.1+cu118", "ephysatlas": "0.9.0", "iblatlas": "1.2.0" }, "method": "gmm", "granularity": "unit", "artifacts": { "autoencoder": "autoencoder.pt", "config": "config.json", "scaler": "shared_latent_scaler.joblib", "gmm": "global_gmm.joblib", "context_transform": "context_transform.joblib", "context_weights": "context_weight_model_bundle.pt", "knn_bank": "knn_bank.npz", "component_features": "component_feature_expectations.npz", "stats": "preprocessing/unit_stats.npz", "split": "split.json", "context": [ "agea_vol_pca.npy", "merfish_vol_pca.npy" ], "results": [ "results/summary.json" ] }, "inputs": { "index": [ "pid", "cluster" ], "columns": [ "x", "y", "z" ], "modalities": [ "waveform", "acg", "stpc" ] }, "outputs": { "kind": "continuous", "columns": [ "depolarisation_slope", "recovery_slope", "repolarisation_slope", "spatial_spread_um", "tip_val", "spike_width_secs", "predepolarisation_width_secs", "spike_amplitude", "peak_to_trough_ratio_log", "polarity" ], "feature_order_sha256": "8536d8c5c99959110df046905224cf91de680434287153131d3ba85eaed968af", "latent_dim": 60 }, "config": { "architecture": { "modalities": [ "waveform", "acg", "stpc" ], "modality_latent_dim": 20, "latent_dim": 60, "gmm_components": 25, "gmm_covariance_type": "full", "knn_k": 20 }, "context": { "n_cell_pcs": 50, "n_gene_pcs": 50, "conditions": "mixture weights, and which training exemplars represent each component in the phenotype readout; component means/covariances are global", "source_model": "int-brain-lab/ea-encoder-channel" }, "preprocessing": { "waveforms": "per-unit max-abs normalized, 20 channels centred on the peak channel", "stats_file": "preprocessing/unit_stats.npz", "mirror_x_to_single_hemisphere": true, "mirror_x_sign": -1.0 }, "readout": { "method": "context_local_members", "description": "the molecular context sets the mixture weights gamma_k(x) and selects which training exemplars represent each component: its members nearest to x in a context key, shrunk toward all of its members -- further as the key leaves the training data, and entirely without context", "key_dim": 3, "key_ridge_alpha": 1000.0, "neighbours": 4096, "shrinkage": 10.0, "off_data_quantile": 0.99, "absent_context": "global member means" } }, "data_source": { "backend": "s3-ibl", "project": "ibl_neuropixel_brainwide_01_2026_W41", "requires_one": true } }