{ "architecture": "AMOR", "format": "custom-pytorch-amor-v1", "model_kwargs": { "d_model": 1024, "n_layer": 24, "d_ff": 1984, "head_dim": 64, "d_state": 128, "d_conv": 4, "expand": 2, "n_groups": 1, "backbone": "gdn", "n_amor_blocks": 3, "residual_mode": "classic", "vocab_size": 128256, "n_heads": 16, "attention_mode": "full_with_mask", "max_seq_len": 4096, "init_alpha": 0.0, "init_threshold": 0.3, "gate_offset": 0.03, "offset_mode": "adaptive", "offset_k": 0.2 }, "tokenizer": "teknium/Llama-3.1-AlternateTokenizer", "tokenizer_revision": "5590cc63198b0f111f4289daf006da0a12044173", "training_context_length": 3072, "tied_weights": { "lm_head.weight": "embed.weight" }, "note": "Use load_model.py; not compatible with Transformers AutoModel or GGUF.", "tokenizer_revision_provenance": "Pinned at release packaging; original training-time revision not recorded." }