DiffuRefill-1B / config.json
Asilarkness's picture
Super-squash branch 'main' using huggingface_hub
a02b918
Raw History Blame Contribute Delete
8.1 kB
{
"architectures": [
"MaskedDiffusionLM"
],
"model_type": "diffurefill",
"_comment": "Not a HuggingFace-loadable config. The reference implementation is big_common.py in this repo; the fields below describe that module exactly.",
"hidden_size": 2048,
"num_hidden_layers": 18,
"num_attention_heads": 16,
"num_key_value_heads": 16,
"head_dim": 128,
"intermediate_size": 5632,
"hidden_act": "silu",
"mlp_type": "swiglu",
"norm_type": "rmsnorm",
"norm_placement": "pre",
"attention_bias": false,
"mlp_bias": false,
"position_encoding": "rope",
"rope_theta": 10000.0,
"rope_applied_to": [
"q",
"k"
],
"max_position_embeddings": 2048,
"attention_mask": "bidirectional",
"attention_impl": "scaled_dot_product_attention; flex_attention with a document block mask when BG_DOCMASK=1",
"vocab_size": 73760,
"tokenizer": "openbmb/MiniCPM4-0.5B",
"tokenizer_vocab_size": 73440,
"mask_token_id": 73759,
"pad_token_id": 73758,
"tie_word_embeddings": true,
"torch_dtype": "bfloat16",
"parameters": {
"total": 1076000000,
"embedding_tied_with_head": 151060480,
"per_block": 51384320,
"per_block_attention": 16777216,
"per_block_mlp": 34603008,
"note": "vocab is padded from the tokenizer's 73440 to 73760 to reserve MASK and PAD and keep the head a round size"
},
"layer_order": [
"tok_embedding (73760 x 2048, tied to output head)",
"18 x Block:",
" RMSNorm -> Linear(2048 -> 6144, no bias) -> split q,k,v -> RoPE(q,k) -> bidirectional SDPA -> Linear(2048 -> 2048, no bias) -> residual add",
" RMSNorm -> [Linear(2048 -> 5632) SiLU] * [Linear(2048 -> 5632)] -> Linear(5632 -> 2048) -> residual add",
"final RMSNorm",
"output head = tok_embedding transposed (tied)"
],
"init": {
"embedding": "normal(0, 0.02)",
"block_linears": "normal(0, 0.02 / sqrt(2 * num_hidden_layers))",
"note": "the head is tied to the embedding, so the embedding scale sets the logit scale"
},
"objective": {
"type": "masked_absorbing_diffusion",
"loss": "cross_entropy on masked positions only",
"mask_rate": "t ~ Uniform(0.15, 1.0) per row, positions masked independently",
"expected_supervised_fraction": 0.575,
"prefix_conditioning_prob": 0.5,
"prefix_note": "on half of rows a random-length clean prefix is never masked, so training sees the prompt-conditioned shape generation uses",
"head_note": "only masked positions are pushed through the 73760-wide head",
"document_mask": "attention is kept inside each packed document (flex_attention block mask)",
"salient_span_masking": {
"rows": 0.25,
"rule": "on a quarter of rows, names and numbers - capitalised words that do not merely open a sentence, and digit runs, merged into multi-word spans - are masked as whole units with probability sqrt(t) >= t; all other positions keep their independent draw",
"why": "masked independently, a two-token name is hidden in full with probability t^2 and the visible half gives it away; hidden as a unit it must be recalled from the rest of the sentence"
},
"span_masking_prob": 0.0,
"span_note": "contiguous span masking (mean 3) was used until step ~91.9k and then removed: independent masking beat it both by one-pass CE and by decoding accuracy on the 37.9M stand. PMI-unit masking, stratified mask rates and complementary masks were also tested there and lost to plain independent masking."
},
"training": {
"optimizer": "AdamW",
"betas": [
0.9,
0.95
],
"eps": 1e-08,
"weight_decay": 0.1,
"grad_clip": 1.0,
"peak_lr": 0.00015,
"warmup_steps": 500,
"schedule": "cosine to 0.1 of peak over 200000 steps",
"sequence_length": 2048,
"tokens_per_optimizer_step": 131072,
"batch_sequences": 64,
"micro_batch": 16,
"dataset": "see dataset_mix",
"distributed": {
"method": "DiLoCo through this repository",
"workers": "3-5 single-GPU workers (RTX PRO 6000 Blackwell, 96 GB), boxes come and go",
"sharding": "each worker takes every Nth document of the base corpus from its own shuffle; the extra sources are read unsharded",
"local_steps_between_merges": 150,
"outer_optimizer": "plain average (momentum 0, lr 1); outer Nesterov was tried and raised held-out CE",
"global_adoption": "theta <- global (plain replacement)",
"guard": "a merge is rejected if it raises held-out masked CE by more than 25%",
"global_adoption_note": "an earlier version kept local drift across a merge (theta += global - last_pushed). Measured on a fixed probe it made the model worse in 60 of 63 adoptions, mean +0.079 CE, against -0.08 to -0.10 for plain replacement: it preserves exactly the disagreeing component that averaging exists to divide by N, and it accumulates.",
"merge": "the merger folds each worker checkpoint into a running fp32 sum as it arrives, waits at most 150 s past the first arrival, and publishes global_b.pt",
"phase": "workers adopt a new global as soon as it appears (checked every 10 steps) and take its step counter, so the whole fleet starts each 150-step interval together"
},
"peak_lr_note": "halved from 3e-4. At 3e-4 a worker drifted 0.076 CE away from the merge point over one 150-step interval, which is more than averaging can reconcile; at 1.5e-4 the drift is 0.013 and the merged model fell from 4.336 to 4.288 on a fixed held-out probe.",
"dataset_mix": {
"note": "documents are drawn by source with these probabilities from step ~96.8k; before that the model saw only the two Ultra-FineWeb configs",
"openbmb/Ultra-FineWeb-L3 (Multi-Style-Synthetic + QA-Synthetic)": 0.56,
"wikimedia/wikipedia 20231101.en": 0.15,
"HuggingFaceTB/finemath finemath-4plus": 0.08,
"HuggingFaceTB/smollm-corpus cosmopedia-v2": 0.08,
"OpenCoder-LLM/opc-annealing-corpus algorithmic_corpus": 0.04,
"OpenCoder-LLM/opc-annealing-corpus synthetic_code_snippet": 0.04,
"Asilarkness/DiffuRefill-facts (Wikipedia leads restated as textbook, lecture, museum-guide ... prose)": 0.05
},
"ema": {
"beta": 0.9,
"file": "global_ema.pt",
"note": "an exponential average of the merged globals; it beats the latest global on held-out CE every round. It is published for use, never used to seed training."
},
"compile": "each block is torch.compile'd in place: 17% faster steps and less memory, same parameter names"
},
"decoding_default": {
"draft_passes_K": 12,
"refill_rounds": 8,
"refill_fraction": 0.2,
"draft_temp_start": 2.2,
"draft_temp_end": 0.9,
"gumbel_tau": 2.5,
"refill_temp": 0.8,
"refill_min_p": 0.1,
"neighbour_ban": 6.0,
"min_commit_gap": 2,
"min_commit_gap_note": "two positions closer than this are never committed in the same pass; co-committed tokens are sampled from their own marginals as if independent, and neighbours are where that is most wrong. Measured on held-out text: adjacent-pair accuracy rises at every pass budget."
},
"status": {
"stage": "pretraining, about half way",
"lineage_step": 100050,
"target_steps": 200000,
"effective_tokens": 46150000000,
"effective_tokens_note": "summed over every worker interval that went into a merge (one interval = 150 steps x 131072 tokens); the per-worker step counter alone undercounts by the number of workers",
"round": 290,
"best_checkpoint": "global_ema.pt",
"note": "a pretraining checkpoint, not an instruction-tuned model; it writes fluent English but its facts are still unreliable"
},
"attention_note": "documents are packed back to back, so without a mask a masked position can be filled from an unrelated neighbouring document. The block-diagonal mask is not a speed cost: measured at the training shape it runs 13.2 ms against 17.5 ms unmasked, because blocks lying in no document are skipped entirely."
}