#!/usr/bin/env python3 """Generate immutable Hugging Face BF16 reference outputs for GGUF checks.""" from __future__ import annotations import argparse import hashlib import json import os import platform from pathlib import Path os.environ.setdefault("HF_HUB_OFFLINE", "1") os.environ.setdefault("TRANSFORMERS_OFFLINE", "1") os.environ.setdefault("TOKENIZERS_PARALLELISM", "false") import torch import transformers from transformers import AutoModelForCausalLM, AutoTokenizer PROMPTS = { "greedy_capitals_12": { "text": ( "The capital of France is Paris. The capital of Germany is Berlin. " "The capital of Japan is" ), "max_new_tokens": 12, }, "greedy_arithmetic_64": { "text": ( "SYSTEMdetailed thinking on<|role_end|>" "HUMANCalculate 17 × 23 and output only the number." "<|role_end|>ASSISTANT\n" ), "max_new_tokens": 64, }, } TOKENIZER_CASES = [ "The capital of France is Paris.", "你好,世界!这是 Ling-3.0-tiny。", "def fibonacci(n: int) -> int:\n return n if n < 2 else fibonacci(n-1) + fibonacci(n-2)", " leading\twhitespace\n\nand trailing ", "emoji: 🧠🚀 café naïve العربية हिन्दी", "SYSTEMdetailed thinking off<|role_end|>HUMANHello<|role_end|>", ] SOURCE_REVISION = "a2ee06c0f2de5b171701aee7f73f70a1da75483b" EXPECTED_WEIGHT_MANIFEST_SHA256 = ( "d8a7cf059fd4b4f2fd7f7d0b1118417be02f7b39a5c428461c027f5d71faff6d" ) EXPECTED_CONTROL_SHA256 = { "chat_template.jinja": "eb6226c94ae38058f875d159f86a206b3a165828c0e7d6bda664ae14667f798a", "config.json": "9750d847957913f665a13c0b5a6537199e33c6f3ec970d9fcb55a0e5076d4012", "configuration_bailing_moe_v3.py": "f2c048966aec8a2f042cfeb1351f74d51a28589b409c55baae7d24e841c1f6c4", "generation_config.json": "64752c5973a55faf4cfc02604c7587c38b090f00013f91191f52362dcc79a4a8", "model.safetensors.index.json": "84ef9fe8ef967eeb0545deb1d23c0ce54e86e6b18fa943a7903d7354a79f9cf9", "modeling_bailing_moe_v3.py": "c2509bf7ac580c262e2581d34d6403aa21682d2e10beb9ad85ad8820a7e33a40", "special_tokens_map.json": "69b63b9f81044ead642d16a5fdc01bcc737dc1183746485c8397ab14d3126614", "tokenizer.json": "40fb9d7d7795b8bd305aeff39ce9963f3f450915b9553f2938e009be9a1fed60", "tokenizer_config.json": "2456b0372956cd3e82f17e33372148b115a94970bfd4878ba5e7e60cd3204f74", } def sha256(path: Path) -> str: digest = hashlib.sha256() with path.open("rb") as stream: for block in iter(lambda: stream.read(1024 * 1024), b""): digest.update(block) return digest.hexdigest() def verify_weight_manifest(model_dir: Path, manifest: Path) -> int: lines = [line for line in manifest.read_text(encoding="utf-8").splitlines() if line] if len(lines) != 32: raise SystemExit(f"expected 32 source shard hashes, got {len(lines)}") for line in lines: expected, filename = line.split(maxsplit=1) actual = sha256(model_dir / filename.strip()) if actual != expected: raise SystemExit(f"source shard hash mismatch: {filename.strip()}") return len(lines) def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("model_dir", type=Path) parser.add_argument("output", type=Path) parser.add_argument("--source-weight-manifest", type=Path, required=True) args = parser.parse_args() if args.model_dir.name != "Ling-3.0-tiny": raise SystemExit("source directory basename must be Ling-3.0-tiny") weight_manifest_sha256 = sha256(args.source_weight_manifest) if weight_manifest_sha256 != EXPECTED_WEIGHT_MANIFEST_SHA256: raise SystemExit( "source weight manifest hash mismatch: " f"expected {EXPECTED_WEIGHT_MANIFEST_SHA256}, " f"got {weight_manifest_sha256}" ) control_hashes = { filename: sha256(args.model_dir / filename) for filename in EXPECTED_CONTROL_SHA256 } if control_hashes != EXPECTED_CONTROL_SHA256: raise SystemExit(f"source control hash mismatch: {control_hashes}") verified_weight_shards = verify_weight_manifest( args.model_dir, args.source_weight_manifest ) torch.manual_seed(1) torch.cuda.manual_seed_all(1) torch.backends.cuda.matmul.allow_tf32 = False torch.backends.cudnn.allow_tf32 = False tokenizer = AutoTokenizer.from_pretrained( args.model_dir, trust_remote_code=True, local_files_only=True, ) model = AutoModelForCausalLM.from_pretrained( args.model_dir, trust_remote_code=True, local_files_only=True, dtype=torch.bfloat16, low_cpu_mem_usage=True, ).eval().to("cuda") result: dict[str, object] = { "source": { "path_basename": args.model_dir.name, "revision": SOURCE_REVISION, "control_file_sha256": control_hashes, "config_sha256": control_hashes["config.json"], "tokenizer_sha256": control_hashes["tokenizer.json"], "verified_weight_shards": verified_weight_shards, "weight_manifest_sha256": weight_manifest_sha256, }, "environment": { "python": platform.python_version(), "torch": torch.__version__, "transformers": transformers.__version__, "cuda": torch.version.cuda, "gpu": torch.cuda.get_device_name(), "dtype": str(next(model.parameters()).dtype), "greedy": True, "tf32": False, "seed": 1, }, "tokenizer_cases": [], "generations": {}, } result["tokenizer_cases"] = [ {"text": text, "token_ids": tokenizer.encode(text, add_special_tokens=False)} for text in TOKENIZER_CASES ] with torch.inference_mode(): for name, case in PROMPTS.items(): prompt = str(case["text"]) encoded = tokenizer(prompt, return_tensors="pt", add_special_tokens=False) encoded = { key: value.to("cuda") for key, value in encoded.items() if key in {"input_ids", "attention_mask"} } output = model.generate( **encoded, max_new_tokens=int(case["max_new_tokens"]), do_sample=False, use_cache=True, pad_token_id=tokenizer.pad_token_id, eos_token_id=tokenizer.eos_token_id, )[0] prompt_tokens = encoded["input_ids"].shape[-1] new_tokens = output[prompt_tokens:].tolist() result["generations"][name] = { "prompt": prompt, "prompt_token_ids": encoded["input_ids"][0].tolist(), "new_token_ids": new_tokens, "new_text": tokenizer.decode(new_tokens, skip_special_tokens=False), "stopped_on_eos": bool(new_tokens and new_tokens[-1] == tokenizer.eos_token_id), } args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_text( json.dumps(result, ensure_ascii=False, indent=2, sort_keys=True) + "\n", encoding="utf-8", ) print(json.dumps(result, ensure_ascii=False, sort_keys=True)) if __name__ == "__main__": main()