Text Generation
Transformers
Safetensors
Korean
English
perdix
custom_code
base-model
differential-attention
polynorm
Perdix-1.1B-Base / config.json
prismdata's picture
Perdix-1.1B-Base: 1.1B base model (30B tokens)
418a411 verified
Raw History Blame Contribute Delete
552 Bytes
{
"architectures": [
"PerdixForCausalLM"
],
"auto_map": {
"AutoConfig": "configuration_perdix.PerdixConfig",
"AutoModelForCausalLM": "modeling_perdix.PerdixForCausalLM"
},
"bos_token_id": 0,
"dim": 2048,
"dtype": "float32",
"eos_token_id": 0,
"ffn_dim": 8192,
"init_std": 0.02,
"max_seq_len": 2048,
"model_type": "perdix",
"n_heads": 16,
"n_layers": 20,
"norm_eps": 1e-05,
"pad_token_id": 1,
"rope_theta": 10000.0,
"tie_word_embeddings": true,
"transformers_version": "5.18.0",
"vocab_size": 49152
}