File size: 1,475 Bytes
eca5751
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
# Nexus Coder Configuration - Medium version v0.3
# ~1B params, pretrain on 4-8 GPU
# Author: Hieu Louis (2026)

model:
  name: "Nexus Coder Medium"
  agent_name: "Nexus"
  author: "Hieu Louis"
  version: "0.3.0-medium"
  github: "mhieuhonda"
  year: "2026"

architecture:
  vocab_size: 32000
  hidden_size: 1536
  num_hidden_layers: 24
  num_attention_heads: 16
  num_kv_heads: 4
  head_dim: 96
  intermediate_size: 4096
  hidden_act: "silu"
  norm_type: "rmsnorm"

moe:
  num_experts: 16
  num_active_experts: 2
  router_aux_loss_coef: 0.001

context:
  max_position_embeddings: 16384
  rotary_emb_base: 10000.0

attention:
  use_flash_attention: true
  use_flash_attention_2: false
  use_alibi: false
  use_sliding_window: true
  sliding_window_size: 2048
  use_qk_norm: true
  mlp_parallel: true

compute:
  use_kv_cache: true
  kv_cache_quantization: null
  gradient_checkpointing: false

params:
  total: "~1.1B"
  active: "~250M"
  expert_utilization: "12.5%"

training:
  learning_rate: 3.0e-4
  weight_decay: 0.01
  warmup_steps: 100
  max_steps: 5000
  per_device_batch_size: 4
  gradient_accumulation_steps: 4
  logging_steps: 10
  save_steps: 500
  max_grad_norm: 1.0
  seed: 42
  use_amp: true

inference:
  max_new_tokens: 200
  temperature: 0.8
  top_k: 50
  top_p: 0.9
  do_sample: true

personality:
  type: "humorous"
  language: "bilingual"

environment:
  python_version: "3.12.13"
  pytorch_version: ">=2.0"
  cuda_required: true
  min_gpu_memory_gb: 16