{ "architectures": [ "MaskedDiffusionLM" ], "model_type": "diffurefill", "_comment": "Not a HuggingFace-loadable config. The reference implementation is big_common.py in this repo; the fields below describe that module exactly.", "hidden_size": 2048, "num_hidden_layers": 18, "num_attention_heads": 16, "num_key_value_heads": 16, "head_dim": 128, "intermediate_size": 5632, "hidden_act": "silu", "mlp_type": "swiglu", "norm_type": "rmsnorm", "norm_placement": "pre", "attention_bias": false, "mlp_bias": false, "position_encoding": "rope", "rope_theta": 10000.0, "rope_applied_to": [ "q", "k" ], "max_position_embeddings": 2048, "attention_mask": "bidirectional", "attention_impl": "scaled_dot_product_attention; flex_attention with a document block mask when BG_DOCMASK=1", "vocab_size": 73760, "tokenizer": "openbmb/MiniCPM4-0.5B", "tokenizer_vocab_size": 73440, "mask_token_id": 73759, "pad_token_id": 73758, "tie_word_embeddings": true, "torch_dtype": "bfloat16", "parameters": { "total": 1076000000, "embedding_tied_with_head": 151060480, "per_block": 51384320, "per_block_attention": 16777216, "per_block_mlp": 34603008, "note": "vocab is padded from the tokenizer's 73440 to 73760 to reserve MASK and PAD and keep the head a round size" }, "layer_order": [ "tok_embedding (73760 x 2048, tied to output head)", "18 x Block:", " RMSNorm -> Linear(2048 -> 6144, no bias) -> split q,k,v -> RoPE(q,k) -> bidirectional SDPA -> Linear(2048 -> 2048, no bias) -> residual add", " RMSNorm -> [Linear(2048 -> 5632) SiLU] * [Linear(2048 -> 5632)] -> Linear(5632 -> 2048) -> residual add", "final RMSNorm", "output head = tok_embedding transposed (tied)" ], "init": { "embedding": "normal(0, 0.02)", "block_linears": "normal(0, 0.02 / sqrt(2 * num_hidden_layers))", "note": "the head is tied to the embedding, so the embedding scale sets the logit scale" }, "objective": { "type": "masked_absorbing_diffusion", "loss": "cross_entropy on masked positions only", "mask_rate": "t ~ Uniform(0.15, 1.0) per row, positions masked independently", "expected_supervised_fraction": 0.575, "prefix_conditioning_prob": 0.5, "prefix_note": "on half of rows a random-length clean prefix is never masked, so training sees the prompt-conditioned shape generation uses", "head_note": "only masked positions are pushed through the 73760-wide head", "document_mask": "attention is kept inside each packed document (flex_attention block mask)", "salient_span_masking": { "rows": 0.25, "rule": "on a quarter of rows, names and numbers - capitalised words that do not merely open a sentence, and digit runs, merged into multi-word spans - are masked as whole units with probability sqrt(t) >= t; all other positions keep their independent draw", "why": "masked independently, a two-token name is hidden in full with probability t^2 and the visible half gives it away; hidden as a unit it must be recalled from the rest of the sentence" }, "span_masking_prob": 0.0, "span_note": "contiguous span masking (mean 3) was used until step ~91.9k and then removed: independent masking beat it both by one-pass CE and by decoding accuracy on the 37.9M stand. PMI-unit masking, stratified mask rates and complementary masks were also tested there and lost to plain independent masking." }, "training": { "optimizer": "AdamW", "betas": [ 0.9, 0.95 ], "eps": 1e-08, "weight_decay": 0.1, "grad_clip": 1.0, "peak_lr": 0.00015, "warmup_steps": 500, "schedule": "cosine to 0.1 of peak over 200000 steps", "sequence_length": 2048, "tokens_per_optimizer_step": 131072, "batch_sequences": 64, "micro_batch": 16, "dataset": "see dataset_mix", "distributed": { "method": "DiLoCo through this repository", "workers": "3-5 single-GPU workers (RTX PRO 6000 Blackwell, 96 GB), boxes come and go", "sharding": "each worker takes every Nth document of the base corpus from its own shuffle; the extra sources are read unsharded", "local_steps_between_merges": 150, "outer_optimizer": "plain average (momentum 0, lr 1); outer Nesterov was tried and raised held-out CE", "global_adoption": "theta <- global (plain replacement)", "guard": "a merge is rejected if it raises held-out masked CE by more than 25%", "global_adoption_note": "an earlier version kept local drift across a merge (theta += global - last_pushed). Measured on a fixed probe it made the model worse in 60 of 63 adoptions, mean +0.079 CE, against -0.08 to -0.10 for plain replacement: it preserves exactly the disagreeing component that averaging exists to divide by N, and it accumulates.", "merge": "the merger folds each worker checkpoint into a running fp32 sum as it arrives, waits at most 150 s past the first arrival, and publishes global_b.pt", "phase": "workers adopt a new global as soon as it appears (checked every 10 steps) and take its step counter, so the whole fleet starts each 150-step interval together" }, "peak_lr_note": "halved from 3e-4. At 3e-4 a worker drifted 0.076 CE away from the merge point over one 150-step interval, which is more than averaging can reconcile; at 1.5e-4 the drift is 0.013 and the merged model fell from 4.336 to 4.288 on a fixed held-out probe.", "dataset_mix": { "note": "documents are drawn by source with these probabilities from step ~96.8k; before that the model saw only the two Ultra-FineWeb configs", "openbmb/Ultra-FineWeb-L3 (Multi-Style-Synthetic + QA-Synthetic)": 0.56, "wikimedia/wikipedia 20231101.en": 0.15, "HuggingFaceTB/finemath finemath-4plus": 0.08, "HuggingFaceTB/smollm-corpus cosmopedia-v2": 0.08, "OpenCoder-LLM/opc-annealing-corpus algorithmic_corpus": 0.04, "OpenCoder-LLM/opc-annealing-corpus synthetic_code_snippet": 0.04, "Asilarkness/DiffuRefill-facts (Wikipedia leads restated as textbook, lecture, museum-guide ... prose)": 0.05 }, "ema": { "beta": 0.9, "file": "global_ema.pt", "note": "an exponential average of the merged globals; it beats the latest global on held-out CE every round. It is published for use, never used to seed training." }, "compile": "each block is torch.compile'd in place: 17% faster steps and less memory, same parameter names" }, "decoding_default": { "draft_passes_K": 12, "refill_rounds": 8, "refill_fraction": 0.2, "draft_temp_start": 2.2, "draft_temp_end": 0.9, "gumbel_tau": 2.5, "refill_temp": 0.8, "refill_min_p": 0.1, "neighbour_ban": 6.0, "min_commit_gap": 2, "min_commit_gap_note": "two positions closer than this are never committed in the same pass; co-committed tokens are sampled from their own marginals as if independent, and neighbours are where that is most wrong. Measured on held-out text: adjacent-pair accuracy rises at every pass budget." }, "status": { "stage": "pretraining, about half way", "lineage_step": 100050, "target_steps": 200000, "effective_tokens": 46150000000, "effective_tokens_note": "summed over every worker interval that went into a merge (one interval = 150 steps x 131072 tokens); the per-worker step counter alone undercounts by the number of workers", "round": 290, "best_checkpoint": "global_ema.pt", "note": "a pretraining checkpoint, not an instruction-tuned model; it writes fluent English but its facts are still unreliable" }, "attention_note": "documents are packed back to back, so without a mask a masked position can be filled from an unrelated neighbouring document. The block-diagonal mask is not a speed cost: measured at the training shape it runs 13.2 ms against 17.5 ms unmasked, because blocks lying in no document are skipped entirely." }