PakMosaic-Small / evals.json
mahwizzzz's picture
Add PakMosaic-Small research-preview weights, training summary, and diagnostic evals
9ae456b verified
Raw History Blame Contribute Delete
2.59 kB
{
"public_name": "PakMosaic-Small",
"run_id": "pakmosaic-small_20260925T090924Z",
"step": 2904769,
"tokens_seen_checkpoint": 2000000068,
"parameter_count": {
"total": 67037952,
"trainable": 67037952
},
"training": {
"step": 2904769,
"tokens_seen": 2000000068,
"token_budget": 2000000000,
"last_loss": 3.107429265975952,
"last_masked_accuracy": 0.4583333432674408,
"mean_loss_last_500": 1.9579948586970568,
"mean_masked_accuracy_last_500": 0.6402293934822082,
"n": 500
},
"diagnostic_eval": {
"checkpoint": "/home/mahwiz/ai-workspace/mahwiz/pakmosaic/runs/pakmosaic-small_20260925T090924Z/final.pt",
"step": 2904769,
"perturbation": {
"label": "DIAGNOSTIC_NOT_PAKBENCH",
"kind": "same_sentence_perturbation",
"cosine": 0.7603878974914551,
"note": "Engineering sanity only. Not a semantic benchmark."
},
"urdu_pairs": {
"label": "DIAGNOSTIC_NOT_PAKBENCH",
"kind": "urdu_pair_sanity",
"pairs": [
{
"tag": "near",
"cosine": 0.9000482559204102
},
{
"tag": "far",
"cosine": 0.6890260577201843
}
]
},
"language_centroids": {
"label": "DIAGNOSTIC_NOT_PAKBENCH",
"kind": "language_clustering_diagnostic",
"pairwise_cosine": {
"urd_Arab__pus_Arab": 0.8904119729995728
},
"note": "Untrained or barely-trained centroids are not language-ID scores."
},
"label": "DIAGNOSTIC_NOT_PAKBENCH"
},
"sota_gate": {
"SOTA_VERIFIED": false,
"checkpoint": "/home/mahwiz/ai-workspace/mahwiz/pakmosaic/runs/pakmosaic-small_20260925T090924Z/final.pt",
"missing_evidence": [
"frozen_evaluation",
"contamination_zero",
"multiple_pakistan_languages",
"contemporary_baselines",
"task_level_results",
"macro",
"low_resource_macro",
"worst_language",
"not_tokenizer_fertility_only"
],
"public_label": "research preview",
"note": "A tokenizer fertility win is not model SOTA. Do not write SOTA on a model card until this gate returns true."
},
"SOTA_VERIFIED": false,
"label": "research preview",
"note": "Diagnostic evals are not PakBench. Weights are inference safetensors only.",
"safetensors": {
"path": "/home/mahwiz/ai-workspace/mahwiz/pakmosaic/release/hf_small_staging/model.safetensors",
"sha256": "209c1aa99ff759ffdad62f38467ff4a8945c7a94397a7440ab815ba9e47529b7"
},
"not_uploaded": [
"FineWeb2",
"HPLT",
"Common Voice",
"optimizer .pt"
]
}