LoopedLM-P2-d4-m / config.json
HuskyDoge's picture
Initial release
5a8873e
Raw History Blame Contribute Delete
3.48 kB
{
"model": {
"apply_attn_gate": false,
"apply_bias_term": true,
"apply_rmsnorm": true,
"arch": "transformer",
"attention_dropout": 0.0,
"attn_act_func": "softmax",
"attn_gate_func": "silu",
"awm_orthogonal_update": false,
"causal_attn_backend": "flash",
"causal_conv_backend": "triton",
"causal_conv_weight_normalization": true,
"causal_conv_width": 4,
"chunk_size": 1024,
"ddp_backend": "fsdp1",
"dense_prelude_diagonal_a_init": 1.0,
"dense_prelude_diagonal_dt_init": 1.0,
"dense_prelude_input_injection": "none",
"dense_prelude_input_layers": "",
"dense_prelude_source_layer": -1,
"dropout": 0.0,
"expert_inter_dim": 0,
"ffn_hidden_dim": 8704,
"fused_block": false,
"fused_output_layer": true,
"head_dim": 64,
"hidden_dropout": 0.0,
"huginn_antithetic_sampling": false,
"huginn_backprop_depth": null,
"huginn_coda_layers": 1,
"huginn_depth_control": false,
"huginn_depth_prior": "none",
"huginn_depth_prior_entropy": 0.0,
"huginn_diagonal_a_init": 1.0,
"huginn_diagonal_dt_init": 1.0,
"huginn_hierarchical_h_cycles": 2,
"huginn_hierarchical_l_cycles": 3,
"huginn_hierarchical_state": "none",
"huginn_input_injection": "diagonal",
"huginn_ortho_projection_eps": 1e-06,
"huginn_poisson_lognormal_max": 64,
"huginn_poisson_lognormal_sigma": 0.5,
"huginn_poisson_lognormal_target_mean": 5.0,
"huginn_prelude_layers": 1,
"huginn_prelude_norm": "none",
"huginn_prelude_orthogonal": false,
"huginn_recurrent_exit_norm": "none",
"huginn_recurrent_layers": null,
"huginn_sampling_scheme": "fixed",
"huginn_split_module_repeats": 1,
"huginn_state_init": "zero",
"init_embed_std": null,
"init_logits_std": null,
"init_mode": "gaussian",
"init_std": 0.02,
"layernorm_eps": 1e-05,
"layernorm_num_groups": 1,
"layerwise_ckpt": false,
"loop_diagonal_a_init": 1.0,
"loop_diagonal_dt_init": 1.0,
"loop_end_layers": null,
"loop_input_injection": "none",
"loop_start_layer": 0,
"loop_times": 1,
"memory_efficient_norm": false,
"model_dim": 3072,
"moe_expert_backend": "sequential",
"moe_permutation_backend": "torch",
"moe_router_bias": false,
"moe_router_bias_update_rate": null,
"moe_router_scaling_factor": null,
"moe_router_score_func": "sigmoid",
"mova_backend": "sequential",
"norm_affine": true,
"num_activated_experts": 0,
"num_activated_values": 0,
"num_dense_layers": null,
"num_experts": 0,
"num_heads": 48,
"num_kv_heads": 12,
"num_layers": 4,
"num_shared_experts": 0,
"num_values": 0,
"output_size": -1,
"qknorm": false,
"recompute_attention": false,
"recompute_awk": false,
"recompute_fc1_out": false,
"recompute_fc3_out": false,
"recompute_logits": false,
"recompute_q": false,
"recompute_router": false,
"recompute_v": false,
"residual_func": "base",
"residual_heads": null,
"rmsnorm_eps": 1e-06,
"rope_base": 1000000.0,
"rope_head_dim": 64,
"sca_backend": "swift",
"scale_emb": false,
"swiglu": true,
"timenorm_backend": "cub",
"timenorm_beta1": 0.999,
"timenorm_beta2": 0.9999,
"timenorm_eps": 1e-05,
"timenorm_num_groups": 32,
"v_head_dim": null,
"vocab_size": 64256
},
"tokenizer": {
"path": "tokenizer",
"type": "huggingface"
}
}