tueminh commited on
Commit
fed91d3
·
verified ·
1 Parent(s): f79236f

Upload folder using huggingface_hub

Browse files
r4sft19y/final/activation_scaler.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"scaling_factor": null}
r4sft19y/final/activations_store_state.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2fecfc6cd213733b84d31f5042774c0968b3c653328e054407b47c1296860244
3
+ size 88
r4sft19y/final/cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"d_in": 2304, "d_sae": 131072, "dtype": "float32", "device": "cuda:2", "apply_b_dec_to_input": false, "normalize_activations": "none", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.39.0", "sae_lens_training_version": "6.39.0", "dataset_path": "monology/pile-uncopyrighted", "hook_name": "blocks.13.hook_resid_post", "model_name": "google/gemma-2-2b", "model_class_name": "HookedTransformer", "hook_head_index": null, "context_size": 128, "seqpos_slice": [null], "model_from_pretrained_kwargs": {"center_writing_weights": false}, "prepend_bos": true, "exclude_special_tokens": false, "sequence_separator_token": "bos", "disable_concat_sequences": false}, "decoder_init_norm": 0.1, "n_experts": 8192, "d_expert": 16, "d_bottleneck": 3, "k_experts": 32, "aux_loss_coefficient": 9e-06, "rescale_acts_by_decoder_norm": true, "threshold_lr": 0.1, "dead_after_n_passes": 1000, "architecture": "smixae"}
r4sft19y/final/runner_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"sae": {"d_in": 2304, "d_sae": 131072, "dtype": "float32", "device": "cpu", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.39.0", "sae_lens_training_version": "6.39.0"}, "decoder_init_norm": 0.1, "n_experts": 8192, "d_expert": 16, "d_bottleneck": 3, "k_experts": 32, "aux_loss_coefficient": 9e-06, "rescale_acts_by_decoder_norm": true, "threshold_lr": 0.1, "dead_after_n_passes": 1000, "architecture": "smixae"}, "model_name": "google/gemma-2-2b", "model_class_name": "HookedTransformer", "hook_name": "blocks.13.hook_resid_post", "hook_eval": "NOT_IN_USE", "hook_head_index": null, "dataset_path": "monology/pile-uncopyrighted", "dataset_trust_remote_code": true, "streaming": true, "is_dataset_tokenized": false, "use_chat_formatting": false, "context_size": 128, "use_cached_activations": false, "cached_activations_path": null, "from_pretrained_path": null, "n_batches_in_buffer": 1024, "training_tokens": 500000000, "store_batch_size_prompts": 128, "seqpos_slice": [null], "disable_concat_sequences": false, "sequence_separator_token": "bos", "activations_mixing_fraction": 0.5, "device": "cuda:2", "act_store_device": "cpu", "seed": 42, "dtype": "bfloat16", "prepend_bos": true, "autocast": true, "autocast_lm": true, "compile_llm": false, "llm_compilation_mode": null, "compile_sae": false, "sae_compilation_mode": null, "train_batch_size_tokens": 8192, "adam_beta1": 0.9, "adam_beta2": 0.999, "lr": 0.0005, "lr_scheduler_name": "constant", "lr_warm_up_steps": 500, "lr_end": 5e-05, "lr_decay_steps": 12207, "n_restart_cycles": 1, "dead_feature_window": 1000, "feature_sampling_window": 2000, "dead_feature_threshold": 1e-08, "n_eval_batches": 10, "eval_batch_size_prompts": null, "logger": {"log_to_wandb": false, "log_activations_store_to_wandb": false, "log_optimizer_state_to_wandb": false, "log_weights_to_wandb": true, "wandb_project": "SMIXAE on Gemma 2-9B, Batch Top K", "wandb_id": null, "run_name": "smixae-131072-LR-0.0005-Tokens-5.000e+08", "wandb_entity": null, "wandb_log_frequency": 30, "eval_every_n_wandb_logs": 5000000}, "n_checkpoints": 49, "checkpoint_path": "/data/caotue/geometry_checkpoints/l13/r4sft19y", "save_final_checkpoint": true, "output_path": "outputs/smixae_l13", "resume_from_checkpoint": null, "verbose": true, "model_kwargs": {}, "model_from_pretrained_kwargs": {"center_writing_weights": false}, "sae_lens_version": "6.39.0", "sae_lens_training_version": "6.39.0", "exclude_special_tokens": false, "n_batches_for_norm_estimate": 100}
r4sft19y/final/sae_weights.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:731e678c3167c3f958f3005d6d430e047fcf53265020c795121418a4c6ce1967
3
+ size 2419664536
r4sft19y/final/sparsity.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4cb1c6b0bc8aff4dd4a53cead9aa8a8267cdc8b0dd5a862e1e54d33e8ae8bb10
3
+ size 524368
r4sft19y/final/trainer_state.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c88e32ecf1afb34f7735d0cf184f813f1af649f54b44bd5b4bf67031dc45ab87
3
+ size 4840253031
r4sft19y/latest/activation_scaler.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"scaling_factor": 0.2592952601867368}
r4sft19y/latest/activations_store_state.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2fecfc6cd213733b84d31f5042774c0968b3c653328e054407b47c1296860244
3
+ size 88
r4sft19y/latest/cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"d_in": 2304, "d_sae": 131072, "dtype": "float32", "device": "cuda:2", "apply_b_dec_to_input": false, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.39.0", "sae_lens_training_version": "6.39.0", "dataset_path": "monology/pile-uncopyrighted", "hook_name": "blocks.13.hook_resid_post", "model_name": "google/gemma-2-2b", "model_class_name": "HookedTransformer", "hook_head_index": null, "context_size": 128, "seqpos_slice": [null], "model_from_pretrained_kwargs": {"center_writing_weights": false}, "prepend_bos": true, "exclude_special_tokens": false, "sequence_separator_token": "bos", "disable_concat_sequences": false}, "decoder_init_norm": 0.1, "n_experts": 8192, "d_expert": 16, "d_bottleneck": 3, "k_experts": 32, "aux_loss_coefficient": 9e-06, "rescale_acts_by_decoder_norm": true, "threshold_lr": 0.1, "dead_after_n_passes": 1000, "architecture": "smixae"}
r4sft19y/latest/runner_cfg.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"sae": {"d_in": 2304, "d_sae": 131072, "dtype": "float32", "device": "cpu", "apply_b_dec_to_input": true, "normalize_activations": "expected_average_only_in", "reshape_activations": "none", "metadata": {"sae_lens_version": "6.39.0", "sae_lens_training_version": "6.39.0"}, "decoder_init_norm": 0.1, "n_experts": 8192, "d_expert": 16, "d_bottleneck": 3, "k_experts": 32, "aux_loss_coefficient": 9e-06, "rescale_acts_by_decoder_norm": true, "threshold_lr": 0.1, "dead_after_n_passes": 1000, "architecture": "smixae"}, "model_name": "google/gemma-2-2b", "model_class_name": "HookedTransformer", "hook_name": "blocks.13.hook_resid_post", "hook_eval": "NOT_IN_USE", "hook_head_index": null, "dataset_path": "monology/pile-uncopyrighted", "dataset_trust_remote_code": true, "streaming": true, "is_dataset_tokenized": false, "use_chat_formatting": false, "context_size": 128, "use_cached_activations": false, "cached_activations_path": null, "from_pretrained_path": null, "n_batches_in_buffer": 1024, "training_tokens": 500000000, "store_batch_size_prompts": 128, "seqpos_slice": [null], "disable_concat_sequences": false, "sequence_separator_token": "bos", "activations_mixing_fraction": 0.5, "device": "cuda:2", "act_store_device": "cpu", "seed": 42, "dtype": "bfloat16", "prepend_bos": true, "autocast": true, "autocast_lm": true, "compile_llm": false, "llm_compilation_mode": null, "compile_sae": false, "sae_compilation_mode": null, "train_batch_size_tokens": 8192, "adam_beta1": 0.9, "adam_beta2": 0.999, "lr": 0.0005, "lr_scheduler_name": "constant", "lr_warm_up_steps": 500, "lr_end": 5e-05, "lr_decay_steps": 12207, "n_restart_cycles": 1, "dead_feature_window": 1000, "feature_sampling_window": 2000, "dead_feature_threshold": 1e-08, "n_eval_batches": 10, "eval_batch_size_prompts": null, "logger": {"log_to_wandb": false, "log_activations_store_to_wandb": false, "log_optimizer_state_to_wandb": false, "log_weights_to_wandb": true, "wandb_project": "SMIXAE on Gemma 2-9B, Batch Top K", "wandb_id": null, "run_name": "smixae-131072-LR-0.0005-Tokens-5.000e+08", "wandb_entity": null, "wandb_log_frequency": 30, "eval_every_n_wandb_logs": 5000000}, "n_checkpoints": 49, "checkpoint_path": "/data/caotue/geometry_checkpoints/l13/r4sft19y", "save_final_checkpoint": true, "output_path": "outputs/smixae_l13", "resume_from_checkpoint": null, "verbose": true, "model_kwargs": {}, "model_from_pretrained_kwargs": {"center_writing_weights": false}, "sae_lens_version": "6.39.0", "sae_lens_training_version": "6.39.0", "exclude_special_tokens": false, "n_batches_for_norm_estimate": 100}
r4sft19y/latest/sae_weights.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d485ce000599034736ccf4d0b9cf2005ce8f5a7aac03afff770669a533e4f4b4
3
+ size 2419664536
r4sft19y/latest/sparsity.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d77c7a515d2a4607072077af030bfef80e0e6b87f1dc64947e10a51289861214
3
+ size 524368
r4sft19y/latest/trainer_state.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:86b954e9db2b0785fc4b43122fbb238211d245e27abda91756be2d050501a98f
3
+ size 4840253031