ea-encoder-unit / ephysatlas_model.json
AlonSaguy's picture
Publish 2026_W41_unit
afe0db2 verified
Raw History Blame Contribute Delete
3.42 kB
{
"model_id": "2026_W41_unit",
"task": "unit-encoding",
"model_class": "UnitAutoencoder",
"vintage": "2026_W41",
"training": {
"random_seed": 0,
"training_size": 572,
"training_units": 56497,
"validation_size": 75,
"validation_units": 7671,
"testing_size": 149,
"testing_units": 15880
},
"environment": {
"python": "3.11.9",
"xgboost": "3.2.0",
"scikit_learn": "1.8.0",
"numpy": "2.3.5",
"pandas": "3.0.5",
"torch": "2.7.1+cu118",
"ephysatlas": "0.9.0",
"iblatlas": "1.2.0"
},
"method": "gmm",
"granularity": "unit",
"artifacts": {
"autoencoder": "autoencoder.pt",
"config": "config.json",
"scaler": "shared_latent_scaler.joblib",
"gmm": "global_gmm.joblib",
"context_transform": "context_transform.joblib",
"context_weights": "context_weight_model_bundle.pt",
"knn_bank": "knn_bank.npz",
"component_features": "component_feature_expectations.npz",
"stats": "preprocessing/unit_stats.npz",
"split": "split.json",
"context": [
"agea_vol_pca.npy",
"merfish_vol_pca.npy"
],
"results": [
"results/summary.json"
]
},
"inputs": {
"index": [
"pid",
"cluster"
],
"columns": [
"x",
"y",
"z"
],
"modalities": [
"waveform",
"acg",
"stpc"
]
},
"outputs": {
"kind": "continuous",
"columns": [
"depolarisation_slope",
"recovery_slope",
"repolarisation_slope",
"spatial_spread_um",
"tip_val",
"spike_width_secs",
"predepolarisation_width_secs",
"spike_amplitude",
"peak_to_trough_ratio_log",
"polarity"
],
"feature_order_sha256": "8536d8c5c99959110df046905224cf91de680434287153131d3ba85eaed968af",
"latent_dim": 60
},
"config": {
"architecture": {
"modalities": [
"waveform",
"acg",
"stpc"
],
"modality_latent_dim": 20,
"latent_dim": 60,
"gmm_components": 25,
"gmm_covariance_type": "full",
"knn_k": 20
},
"context": {
"n_cell_pcs": 50,
"n_gene_pcs": 50,
"conditions": "mixture weights, and which training exemplars represent each component in the phenotype readout; component means/covariances are global",
"source_model": "int-brain-lab/ea-encoder-channel"
},
"preprocessing": {
"waveforms": "per-unit max-abs normalized, 20 channels centred on the peak channel",
"stats_file": "preprocessing/unit_stats.npz",
"mirror_x_to_single_hemisphere": true,
"mirror_x_sign": -1.0
},
"readout": {
"method": "context_local_members",
"description": "the molecular context sets the mixture weights gamma_k(x) and selects which training exemplars represent each component: its members nearest to x in a context key, shrunk toward all of its members -- further as the key leaves the training data, and entirely without context",
"key_dim": 3,
"key_ridge_alpha": 1000.0,
"neighbours": 4096,
"shrinkage": 10.0,
"off_data_quantile": 0.99,
"absent_context": "global member means"
}
},
"data_source": {
"backend": "s3-ibl",
"project": "ibl_neuropixel_brainwide_01_2026_W41",
"requires_one": true
}
}