File size: 2,161 Bytes
6b96146 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 | from transformers import PretrainedConfig
class LayaConfig(PretrainedConfig):
model_type = "laya"
def __init__(
self,
vocab_size: int = 256000,
hidden_size: int = 768,
intermediate_size: int = 1152,
num_hidden_layers: int = 22,
num_attention_heads: int = 12,
hidden_activation: str = "gelu",
norm_eps: float = 1e-5,
norm_bias: bool = False,
attention_bias: bool = False,
mlp_bias: bool = False,
rope_theta: float = 160000.0,
local_attention: int = 128,
global_attn_every_n_layers: int = 3,
max_position_embeddings: int = 8192,
pad_token_id: int = 0,
cls_token_id: int = 1,
sep_token_id: int = 1,
bos_token_id: int = 2,
mask_token_id: int = 4,
head_layers: int = 2,
head_ff_dim: int = 3072,
n_act: int = 2,
num_question_types: int = 3,
max_len: int = 1024,
head_max_len: int = 256,
**kwargs
):
super().__init__(
pad_token_id=pad_token_id,
bos_token_id=bos_token_id,
sep_token_id=sep_token_id,
**kwargs
)
self.vocab_size = vocab_size
self.hidden_size = hidden_size
self.intermediate_size = intermediate_size
self.num_hidden_layers = num_hidden_layers
self.num_attention_heads = num_attention_heads
self.hidden_activation = hidden_activation
self.norm_eps = norm_eps
self.norm_bias = norm_bias
self.attention_bias = attention_bias
self.mlp_bias = mlp_bias
self.rope_theta = rope_theta
self.local_attention = local_attention
self.global_attn_every_n_layers = global_attn_every_n_layers
self.max_position_embeddings = max_position_embeddings
self.cls_token_id = cls_token_id
self.mask_token_id = mask_token_id
self.head_layers = head_layers
self.head_ff_dim = head_ff_dim
self.n_act = n_act
self.num_question_types = num_question_types
self.max_len = max_len
self.head_max_len = head_max_len
|