Download config.json from Asilarkness/DiffuRefill-1B: direct link, hf CLI and curl.
- Browser
- Download file 8.1 kB
-
https://huggingface.co/Asilarkness/DiffuRefill-1B/resolve/main/config.json
- Command line
-
hf download hf://Asilarkness/DiffuRefill-1B/config.json
-
curl -L -o config.json https://huggingface.co/Asilarkness/DiffuRefill-1B/resolve/main/config.json
8.1 kB
| { | |
| "architectures": [ | |
| "MaskedDiffusionLM" | |
| ], | |
| "model_type": "diffurefill", | |
| "_comment": "Not a HuggingFace-loadable config. The reference implementation is big_common.py in this repo; the fields below describe that module exactly.", | |
| "hidden_size": 2048, | |
| "num_hidden_layers": 18, | |
| "num_attention_heads": 16, | |
| "num_key_value_heads": 16, | |
| "head_dim": 128, | |
| "intermediate_size": 5632, | |
| "hidden_act": "silu", | |
| "mlp_type": "swiglu", | |
| "norm_type": "rmsnorm", | |
| "norm_placement": "pre", | |
| "attention_bias": false, | |
| "mlp_bias": false, | |
| "position_encoding": "rope", | |
| "rope_theta": 10000.0, | |
| "rope_applied_to": [ | |
| "q", | |
| "k" | |
| ], | |
| "max_position_embeddings": 2048, | |
| "attention_mask": "bidirectional", | |
| "attention_impl": "scaled_dot_product_attention; flex_attention with a document block mask when BG_DOCMASK=1", | |
| "vocab_size": 73760, | |
| "tokenizer": "openbmb/MiniCPM4-0.5B", | |
| "tokenizer_vocab_size": 73440, | |
| "mask_token_id": 73759, | |
| "pad_token_id": 73758, | |
| "tie_word_embeddings": true, | |
| "torch_dtype": "bfloat16", | |
| "parameters": { | |
| "total": 1076000000, | |
| "embedding_tied_with_head": 151060480, | |
| "per_block": 51384320, | |
| "per_block_attention": 16777216, | |
| "per_block_mlp": 34603008, | |
| "note": "vocab is padded from the tokenizer's 73440 to 73760 to reserve MASK and PAD and keep the head a round size" | |
| }, | |
| "layer_order": [ | |
| "tok_embedding (73760 x 2048, tied to output head)", | |
| "18 x Block:", | |
| " RMSNorm -> Linear(2048 -> 6144, no bias) -> split q,k,v -> RoPE(q,k) -> bidirectional SDPA -> Linear(2048 -> 2048, no bias) -> residual add", | |
| " RMSNorm -> [Linear(2048 -> 5632) SiLU] * [Linear(2048 -> 5632)] -> Linear(5632 -> 2048) -> residual add", | |
| "final RMSNorm", | |
| "output head = tok_embedding transposed (tied)" | |
| ], | |
| "init": { | |
| "embedding": "normal(0, 0.02)", | |
| "block_linears": "normal(0, 0.02 / sqrt(2 * num_hidden_layers))", | |
| "note": "the head is tied to the embedding, so the embedding scale sets the logit scale" | |
| }, | |
| "objective": { | |
| "type": "masked_absorbing_diffusion", | |
| "loss": "cross_entropy on masked positions only", | |
| "mask_rate": "t ~ Uniform(0.15, 1.0) per row, positions masked independently", | |
| "expected_supervised_fraction": 0.575, | |
| "prefix_conditioning_prob": 0.5, | |
| "prefix_note": "on half of rows a random-length clean prefix is never masked, so training sees the prompt-conditioned shape generation uses", | |
| "head_note": "only masked positions are pushed through the 73760-wide head", | |
| "document_mask": "attention is kept inside each packed document (flex_attention block mask)", | |
| "salient_span_masking": { | |
| "rows": 0.25, | |
| "rule": "on a quarter of rows, names and numbers - capitalised words that do not merely open a sentence, and digit runs, merged into multi-word spans - are masked as whole units with probability sqrt(t) >= t; all other positions keep their independent draw", | |
| "why": "masked independently, a two-token name is hidden in full with probability t^2 and the visible half gives it away; hidden as a unit it must be recalled from the rest of the sentence" | |
| }, | |
| "span_masking_prob": 0.0, | |
| "span_note": "contiguous span masking (mean 3) was used until step ~91.9k and then removed: independent masking beat it both by one-pass CE and by decoding accuracy on the 37.9M stand. PMI-unit masking, stratified mask rates and complementary masks were also tested there and lost to plain independent masking." | |
| }, | |
| "training": { | |
| "optimizer": "AdamW", | |
| "betas": [ | |
| 0.9, | |
| 0.95 | |
| ], | |
| "eps": 1e-08, | |
| "weight_decay": 0.1, | |
| "grad_clip": 1.0, | |
| "peak_lr": 0.00015, | |
| "warmup_steps": 500, | |
| "schedule": "cosine to 0.1 of peak over 200000 steps", | |
| "sequence_length": 2048, | |
| "tokens_per_optimizer_step": 131072, | |
| "batch_sequences": 64, | |
| "micro_batch": 16, | |
| "dataset": "see dataset_mix", | |
| "distributed": { | |
| "method": "DiLoCo through this repository", | |
| "workers": "3-5 single-GPU workers (RTX PRO 6000 Blackwell, 96 GB), boxes come and go", | |
| "sharding": "each worker takes every Nth document of the base corpus from its own shuffle; the extra sources are read unsharded", | |
| "local_steps_between_merges": 150, | |
| "outer_optimizer": "plain average (momentum 0, lr 1); outer Nesterov was tried and raised held-out CE", | |
| "global_adoption": "theta <- global (plain replacement)", | |
| "guard": "a merge is rejected if it raises held-out masked CE by more than 25%", | |
| "global_adoption_note": "an earlier version kept local drift across a merge (theta += global - last_pushed). Measured on a fixed probe it made the model worse in 60 of 63 adoptions, mean +0.079 CE, against -0.08 to -0.10 for plain replacement: it preserves exactly the disagreeing component that averaging exists to divide by N, and it accumulates.", | |
| "merge": "the merger folds each worker checkpoint into a running fp32 sum as it arrives, waits at most 150 s past the first arrival, and publishes global_b.pt", | |
| "phase": "workers adopt a new global as soon as it appears (checked every 10 steps) and take its step counter, so the whole fleet starts each 150-step interval together" | |
| }, | |
| "peak_lr_note": "halved from 3e-4. At 3e-4 a worker drifted 0.076 CE away from the merge point over one 150-step interval, which is more than averaging can reconcile; at 1.5e-4 the drift is 0.013 and the merged model fell from 4.336 to 4.288 on a fixed held-out probe.", | |
| "dataset_mix": { | |
| "note": "documents are drawn by source with these probabilities from step ~96.8k; before that the model saw only the two Ultra-FineWeb configs", | |
| "openbmb/Ultra-FineWeb-L3 (Multi-Style-Synthetic + QA-Synthetic)": 0.56, | |
| "wikimedia/wikipedia 20231101.en": 0.15, | |
| "HuggingFaceTB/finemath finemath-4plus": 0.08, | |
| "HuggingFaceTB/smollm-corpus cosmopedia-v2": 0.08, | |
| "OpenCoder-LLM/opc-annealing-corpus algorithmic_corpus": 0.04, | |
| "OpenCoder-LLM/opc-annealing-corpus synthetic_code_snippet": 0.04, | |
| "Asilarkness/DiffuRefill-facts (Wikipedia leads restated as textbook, lecture, museum-guide ... prose)": 0.05 | |
| }, | |
| "ema": { | |
| "beta": 0.9, | |
| "file": "global_ema.pt", | |
| "note": "an exponential average of the merged globals; it beats the latest global on held-out CE every round. It is published for use, never used to seed training." | |
| }, | |
| "compile": "each block is torch.compile'd in place: 17% faster steps and less memory, same parameter names" | |
| }, | |
| "decoding_default": { | |
| "draft_passes_K": 12, | |
| "refill_rounds": 8, | |
| "refill_fraction": 0.2, | |
| "draft_temp_start": 2.2, | |
| "draft_temp_end": 0.9, | |
| "gumbel_tau": 2.5, | |
| "refill_temp": 0.8, | |
| "refill_min_p": 0.1, | |
| "neighbour_ban": 6.0, | |
| "min_commit_gap": 2, | |
| "min_commit_gap_note": "two positions closer than this are never committed in the same pass; co-committed tokens are sampled from their own marginals as if independent, and neighbours are where that is most wrong. Measured on held-out text: adjacent-pair accuracy rises at every pass budget." | |
| }, | |
| "status": { | |
| "stage": "pretraining, about half way", | |
| "lineage_step": 100050, | |
| "target_steps": 200000, | |
| "effective_tokens": 46150000000, | |
| "effective_tokens_note": "summed over every worker interval that went into a merge (one interval = 150 steps x 131072 tokens); the per-worker step counter alone undercounts by the number of workers", | |
| "round": 290, | |
| "best_checkpoint": "global_ema.pt", | |
| "note": "a pretraining checkpoint, not an instruction-tuned model; it writes fluent English but its facts are still unreliable" | |
| }, | |
| "attention_note": "documents are packed back to back, so without a mask a masked position can be filled from an unrelated neighbouring document. The block-diagonal mask is not a speed cost: measured at the training shape it runs 13.2 ms against 17.5 ms unmasked, because blocks lying in no document are skipped entirely." | |
| } |