"""Rebuild parameter/cost scenarios and capture primary-source revisions.""" from pathlib import Path from datetime import datetime, timezone import importlib.metadata import json import platform import subprocess import sys import urllib.request sys.path.insert(0, str(Path(__file__).resolve().parents[1])) from nexora.compute import Estimate, kv_cache_bytes, topology from huggingface_hub import HfApi def moe(layers, hidden, intermediate, experts, selected, shared, heads, kv, vocab=131072): dim = hidden//heads attention = layers*(2*hidden*hidden + 2*hidden*kv*dim) one_expert = 3*hidden*intermediate expert_total = layers*(experts+shared)*one_expert routing = layers*hidden*experts norms = (2*layers+1)*hidden embedding = vocab*hidden total = attention+expert_total+routing+norms+embedding active = attention+layers*(selected+shared)*one_expert+routing+norms+embedding return {"layers": layers, "hidden_size": hidden, "expert_intermediate": intermediate, "experts": experts, "selected": selected, "shared": shared, "heads": heads, "kv_heads": kv, "head_dim": dim, "vocab": vocab, "attention_parameters": attention, "expert_parameters": expert_total, "router_parameters": routing, "norm_parameters": norms, "tied_embedding_parameters": embedding, "total_parameters": total, "active_parameters_upper_convention": active, "active_convention": "Counts full tied embedding/output matrix, all attention, router and selected/shared FFNs; actual sparse embedding lookup uses fewer elements", "kv_cache_GB_batch1": {str(n): kv_cache_bytes(layers, kv, dim, n)/1e9 for n in [32768, 65536, 131072, 262144]}} def main(): out = Path("reports") out.mkdir(exist_ok=True) proposals = {"B_custom_120B_class": moe(49, 3072, 2048, 128, 4, 1, 24, 8), "C_custom_larger_MoE": moe(80, 4096, 1536, 160, 4, 1, 32, 8)} cases = {"A_dense_120B": (120e9, 120e9, 2.4e12, 1024), "B_custom_120B_class": (proposals["B_custom_120B_class"]["total_parameters"], proposals["B_custom_120B_class"]["active_parameters_upper_convention"], 2.4e12, 256), "C_custom_larger_MoE": (proposals["C_custom_larger_MoE"]["total_parameters"], proposals["C_custom_larger_MoE"]["active_parameters_upper_convention"], 3e12, 512), "D_30B_continue": (30.5e9, 3.3e9, 10e9, 8), "E_4B_continue": (4e9, 4e9, 1e9, 8)} estimates = {} for name, args in cases.items(): estimates[name] = {label: Estimate(*args, mfu=mfu, gpu_hour_usd=rate).calculate() for label, mfu, rate in [("low_cost", .5, 2), ("expected", .35, 3), ("high_cost", .2, 5)]} report = {"assumptions": "H100 SXM BF16 dense peak 989 TFLOP/s planning scenario; MFU .2/.35/.5 not measured; price $2/$3/$5 per GPU-hour are editable hypothetical inputs, not vendor quotes; excludes retries/people/tax/storage/power", "architectures": proposals, "estimates": estimates, "topologies": [topology(1, 8, 1, 1, 1, 8), topology(8, 8, 2, 2, 1, 16, 8), topology(128, 8, 8, 8, 2, 8)], "storage_formula": "3 retained checkpoints * checkpoint_GB + 2 replicated copies of token shards + raw corpus + intermediate dedup data; add validation/holdouts and free-space headroom"} (out / "compute.json").write_text(json.dumps(report, indent=2)) env = {"date_utc": datetime.now(timezone.utc).isoformat(), "os": platform.system(), "python": platform.python_version(), "packages": {p: importlib.metadata.version(p) for p in ["torch", "numpy", "transformers", "safetensors", "huggingface_hub", "jsonschema", "pytest"]}, "hardware": {"cpu": "AMD Ryzen 7 7435HS", "cores": 8, "ram_bytes": 16989736960, "gpu": "NVIDIA RTX 3050 Laptop", "vram_mib": 4096, "cuda_in_installed_torch": False, "free_workspace_drive_bytes_at_discovery": 322162003968}, "cluster": "No allocated cluster discovered", "cloud": "HF authentication available; no paid jobs launched", "budget": "No spending budget specified; local execution only"} (out / "environment.json").write_text(json.dumps(env, indent=2)) sources = [] api = HfApi() for repo in ["Qwen/Qwen3-0.6B", "Qwen/Qwen3.5-0.8B", "Qwen/Qwen3.5-4B", "Qwen/Qwen3-30B-A3B-Instruct-2507"]: info = api.model_info(repo) url = f"https://huggingface.co/{repo}/resolve/{info.sha}/config.json" with urllib.request.urlopen(url, timeout=30) as r: config = json.load(r) sources.append({"repo": repo, "revision": info.sha, "license": info.card_data.get("license") if info.card_data else None, "config_url": url, "card_url": f"https://huggingface.co/{repo}/blob/{info.sha}/README.md", "config": config}) (out / "sources.json").write_text(json.dumps({"accessed_utc": env["date_utc"], "models": sources}, indent=2)) for name, values in estimates.items(): e = values["expected"] print(name, "total_B", e["total_parameters"]/1e9, "active_B", e["active_parameters"]/1e9, "hours", round(e["hours"], 1), "USD", round(e["compute_usd"])) if __name__ == "__main__": main()