File size: 8,098 Bytes
a02b918 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 | {
"architectures": [
"MaskedDiffusionLM"
],
"model_type": "diffurefill",
"_comment": "Not a HuggingFace-loadable config. The reference implementation is big_common.py in this repo; the fields below describe that module exactly.",
"hidden_size": 2048,
"num_hidden_layers": 18,
"num_attention_heads": 16,
"num_key_value_heads": 16,
"head_dim": 128,
"intermediate_size": 5632,
"hidden_act": "silu",
"mlp_type": "swiglu",
"norm_type": "rmsnorm",
"norm_placement": "pre",
"attention_bias": false,
"mlp_bias": false,
"position_encoding": "rope",
"rope_theta": 10000.0,
"rope_applied_to": [
"q",
"k"
],
"max_position_embeddings": 2048,
"attention_mask": "bidirectional",
"attention_impl": "scaled_dot_product_attention; flex_attention with a document block mask when BG_DOCMASK=1",
"vocab_size": 73760,
"tokenizer": "openbmb/MiniCPM4-0.5B",
"tokenizer_vocab_size": 73440,
"mask_token_id": 73759,
"pad_token_id": 73758,
"tie_word_embeddings": true,
"torch_dtype": "bfloat16",
"parameters": {
"total": 1076000000,
"embedding_tied_with_head": 151060480,
"per_block": 51384320,
"per_block_attention": 16777216,
"per_block_mlp": 34603008,
"note": "vocab is padded from the tokenizer's 73440 to 73760 to reserve MASK and PAD and keep the head a round size"
},
"layer_order": [
"tok_embedding (73760 x 2048, tied to output head)",
"18 x Block:",
" RMSNorm -> Linear(2048 -> 6144, no bias) -> split q,k,v -> RoPE(q,k) -> bidirectional SDPA -> Linear(2048 -> 2048, no bias) -> residual add",
" RMSNorm -> [Linear(2048 -> 5632) SiLU] * [Linear(2048 -> 5632)] -> Linear(5632 -> 2048) -> residual add",
"final RMSNorm",
"output head = tok_embedding transposed (tied)"
],
"init": {
"embedding": "normal(0, 0.02)",
"block_linears": "normal(0, 0.02 / sqrt(2 * num_hidden_layers))",
"note": "the head is tied to the embedding, so the embedding scale sets the logit scale"
},
"objective": {
"type": "masked_absorbing_diffusion",
"loss": "cross_entropy on masked positions only",
"mask_rate": "t ~ Uniform(0.15, 1.0) per row, positions masked independently",
"expected_supervised_fraction": 0.575,
"prefix_conditioning_prob": 0.5,
"prefix_note": "on half of rows a random-length clean prefix is never masked, so training sees the prompt-conditioned shape generation uses",
"head_note": "only masked positions are pushed through the 73760-wide head",
"document_mask": "attention is kept inside each packed document (flex_attention block mask)",
"salient_span_masking": {
"rows": 0.25,
"rule": "on a quarter of rows, names and numbers - capitalised words that do not merely open a sentence, and digit runs, merged into multi-word spans - are masked as whole units with probability sqrt(t) >= t; all other positions keep their independent draw",
"why": "masked independently, a two-token name is hidden in full with probability t^2 and the visible half gives it away; hidden as a unit it must be recalled from the rest of the sentence"
},
"span_masking_prob": 0.0,
"span_note": "contiguous span masking (mean 3) was used until step ~91.9k and then removed: independent masking beat it both by one-pass CE and by decoding accuracy on the 37.9M stand. PMI-unit masking, stratified mask rates and complementary masks were also tested there and lost to plain independent masking."
},
"training": {
"optimizer": "AdamW",
"betas": [
0.9,
0.95
],
"eps": 1e-08,
"weight_decay": 0.1,
"grad_clip": 1.0,
"peak_lr": 0.00015,
"warmup_steps": 500,
"schedule": "cosine to 0.1 of peak over 200000 steps",
"sequence_length": 2048,
"tokens_per_optimizer_step": 131072,
"batch_sequences": 64,
"micro_batch": 16,
"dataset": "see dataset_mix",
"distributed": {
"method": "DiLoCo through this repository",
"workers": "3-5 single-GPU workers (RTX PRO 6000 Blackwell, 96 GB), boxes come and go",
"sharding": "each worker takes every Nth document of the base corpus from its own shuffle; the extra sources are read unsharded",
"local_steps_between_merges": 150,
"outer_optimizer": "plain average (momentum 0, lr 1); outer Nesterov was tried and raised held-out CE",
"global_adoption": "theta <- global (plain replacement)",
"guard": "a merge is rejected if it raises held-out masked CE by more than 25%",
"global_adoption_note": "an earlier version kept local drift across a merge (theta += global - last_pushed). Measured on a fixed probe it made the model worse in 60 of 63 adoptions, mean +0.079 CE, against -0.08 to -0.10 for plain replacement: it preserves exactly the disagreeing component that averaging exists to divide by N, and it accumulates.",
"merge": "the merger folds each worker checkpoint into a running fp32 sum as it arrives, waits at most 150 s past the first arrival, and publishes global_b.pt",
"phase": "workers adopt a new global as soon as it appears (checked every 10 steps) and take its step counter, so the whole fleet starts each 150-step interval together"
},
"peak_lr_note": "halved from 3e-4. At 3e-4 a worker drifted 0.076 CE away from the merge point over one 150-step interval, which is more than averaging can reconcile; at 1.5e-4 the drift is 0.013 and the merged model fell from 4.336 to 4.288 on a fixed held-out probe.",
"dataset_mix": {
"note": "documents are drawn by source with these probabilities from step ~96.8k; before that the model saw only the two Ultra-FineWeb configs",
"openbmb/Ultra-FineWeb-L3 (Multi-Style-Synthetic + QA-Synthetic)": 0.56,
"wikimedia/wikipedia 20231101.en": 0.15,
"HuggingFaceTB/finemath finemath-4plus": 0.08,
"HuggingFaceTB/smollm-corpus cosmopedia-v2": 0.08,
"OpenCoder-LLM/opc-annealing-corpus algorithmic_corpus": 0.04,
"OpenCoder-LLM/opc-annealing-corpus synthetic_code_snippet": 0.04,
"Asilarkness/DiffuRefill-facts (Wikipedia leads restated as textbook, lecture, museum-guide ... prose)": 0.05
},
"ema": {
"beta": 0.9,
"file": "global_ema.pt",
"note": "an exponential average of the merged globals; it beats the latest global on held-out CE every round. It is published for use, never used to seed training."
},
"compile": "each block is torch.compile'd in place: 17% faster steps and less memory, same parameter names"
},
"decoding_default": {
"draft_passes_K": 12,
"refill_rounds": 8,
"refill_fraction": 0.2,
"draft_temp_start": 2.2,
"draft_temp_end": 0.9,
"gumbel_tau": 2.5,
"refill_temp": 0.8,
"refill_min_p": 0.1,
"neighbour_ban": 6.0,
"min_commit_gap": 2,
"min_commit_gap_note": "two positions closer than this are never committed in the same pass; co-committed tokens are sampled from their own marginals as if independent, and neighbours are where that is most wrong. Measured on held-out text: adjacent-pair accuracy rises at every pass budget."
},
"status": {
"stage": "pretraining, about half way",
"lineage_step": 100050,
"target_steps": 200000,
"effective_tokens": 46150000000,
"effective_tokens_note": "summed over every worker interval that went into a merge (one interval = 150 steps x 131072 tokens); the per-worker step counter alone undercounts by the number of workers",
"round": 290,
"best_checkpoint": "global_ema.pt",
"note": "a pretraining checkpoint, not an instruction-tuned model; it writes fluent English but its facts are still unreliable"
},
"attention_note": "documents are packed back to back, so without a mask a masked position can be filled from an unrelated neighbouring document. The block-diagonal mask is not a speed cost: measured at the training shape it runs 13.2 ms against 17.5 ms unmasked, because blocks lying in no document are skipped entirely."
} |