"""Erk-Linear yapilandirmasi — %20-lineer hibrit (8/40 dikkat katmani Gated DeltaNet).""" from transformers import PretrainedConfig class ErkLinearConfig(PretrainedConfig): """Hibridin kendisi bir govde tasimaz; govde `base_model`'den yuklenir. Bu yapilandirma yalnizca hangi katmanlarin lineerlestirildigini ve Gated DeltaNet modullerinin nasil kurulacagini tanimlar. Agirliklar iki kaynaktan gelir: - govde (32 softmax katmani + gomme/LM basi) : `base_model` deposundan - 8 GDN katmani : bu deponun gdn_weights.safetensors """ model_type = "erk_linear" def __init__( self, base_model: str = "ecloudtech/Erk-14B", gdn_layers=None, hidden_size: int = 5120, num_hidden_layers: int = 40, gdn_head_dim: int = 128, gdn_num_heads: int = 40, gdn_use_gate: bool = True, gdn_use_short_conv: bool = True, gdn_mode: str = "chunk", gdn_weights_file: str = "gdn_weights.safetensors", **kwargs, ): self.base_model = base_model self.gdn_layers = list(gdn_layers) if gdn_layers is not None else [1, 3, 5, 7, 10, 36, 38, 39] self.hidden_size = hidden_size self.num_hidden_layers = num_hidden_layers self.gdn_head_dim = gdn_head_dim self.gdn_num_heads = gdn_num_heads self.gdn_use_gate = gdn_use_gate self.gdn_use_short_conv = gdn_use_short_conv self.gdn_mode = gdn_mode self.gdn_weights_file = gdn_weights_file super().__init__(**kwargs) @property def linear_ratio(self) -> float: """Lineerlestirilen dikkat katmanlarinin orani (8/40 = 0.20).""" return len(self.gdn_layers) / float(self.num_hidden_layers)