Spaces:
Sleeping
Sleeping
Download scripts/machine_report.py from spitfire4794/test1111111: direct link, hf CLI and curl.
- Browser
- Download file 20.2 kB
-
https://huggingface.co/spaces/spitfire4794/test1111111/resolve/main/scripts/machine_report.py
- Command line
-
hf download hf://spaces/spitfire4794/test1111111/scripts/machine_report.py
-
curl -L -o machine_report.py https://huggingface.co/spaces/spitfire4794/test1111111/resolve/main/scripts/machine_report.py
20.2 kB
| #!/usr/bin/env python | |
| """Machine capability report for deciding where CISM would run best. | |
| Stdlib-only by default; NumPy (if installed) enables measured bandwidth and | |
| sync probes. Works on Windows, Linux, and macOS without admin rights. Run: | |
| python scripts/machine_report.py # human-readable report | |
| python scripts/machine_report.py --json # machine-readable | |
| python scripts/machine_report.py --fast # smaller/faster probes | |
| Interpretation guide is printed at the end: cache sizes set which model | |
| footprints can be cache-resident, the bandwidth ladder shows the measured | |
| cache-to-RAM multiplier on this machine, and the sync probe estimates the | |
| per-barrier cost that penalizes multithreaded sub-millisecond decoding. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| import os | |
| import platform | |
| import statistics | |
| import sys | |
| import threading | |
| import time | |
| def collect_platform() -> dict: | |
| info = { | |
| "os": platform.system(), | |
| "os_release": platform.release(), | |
| "python": platform.python_version(), | |
| "machine": platform.machine(), | |
| } | |
| if info["os"] == "Windows": | |
| try: | |
| import winreg | |
| with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, | |
| r"HARDWARE\DESCRIPTION\System\CentralProcessor\0") as key: | |
| info["cpu_model"] = winreg.QueryValueEx(key, "ProcessorNameString")[0] | |
| except OSError: | |
| info["cpu_model"] = platform.processor() | |
| elif info["os"] == "Linux": | |
| for line in _read_file("/proc/cpuinfo").splitlines(): | |
| if line.lower().startswith("model name"): | |
| info["cpu_model"] = line.split(":", 1)[1].strip() | |
| break | |
| else: | |
| info["cpu_model"] = _sysctl("machdep.cpu.brand_string") or platform.processor() | |
| return info | |
| def collect_cpu_topology() -> dict: | |
| info: dict = {} | |
| system = platform.system() | |
| if system == "Windows": | |
| try: | |
| info["logical_processors"] = os.cpu_count() | |
| result = _powershell( | |
| "Get-CimInstance Win32_Processor | Select-Object -First 1 " | |
| "NumberOfCores,NumberOfLogicalProcessors,L2CacheSize,L3CacheSize,MaxClockSpeed " | |
| "| ConvertTo-Json -Compress") | |
| import json as _json | |
| data = _json.loads(result) | |
| info["physical_cores"] = int(data["NumberOfCores"]) | |
| info["logical_processors"] = int(data["NumberOfLogicalProcessors"]) | |
| if data.get("L2CacheSize"): | |
| info["l2_per_package_kb"] = int(data["L2CacheSize"]) | |
| if data.get("L3CacheSize"): | |
| info["l3_total_kb"] = int(data["L3CacheSize"]) | |
| if data.get("MaxClockSpeed"): | |
| info["max_clock_mhz"] = int(data["MaxClockSpeed"]) | |
| except Exception as error: | |
| info["windows_topology_error"] = str(error) | |
| elif system == "Linux": | |
| info["logical_processors"] = os.cpu_count() | |
| cpuinfo = _read_file("/proc/cpuinfo") | |
| cores = {line.split(":", 1)[1].strip() for line in cpuinfo.splitlines() | |
| if line.strip().startswith("core id")} | |
| physical_ids = {line.split(":", 1)[1].strip() for line in cpuinfo.splitlines() | |
| if line.strip().startswith("physical id")} | |
| info["physical_cores"] = len(cores) * max(len(physical_ids), 1) if cores else None | |
| info["max_clock_mhz"] = max((float(line.split(":", 1)[1]) for line in cpuinfo.splitlines() | |
| if line.strip().startswith("cpu MHz")), default=None) | |
| info["cache"] = _cache_linux() | |
| else: | |
| info["logical_processors"] = os.cpu_count() | |
| info["cache"] = _cache_macos() | |
| if system == "Windows" and "cache" not in info: | |
| try: | |
| info["cache"] = _cache_windows() | |
| except Exception as error: | |
| info["cache_error"] = str(error) | |
| return info | |
| def _read_file(path: str) -> str: | |
| try: | |
| with open(path, encoding="utf-8", errors="replace") as handle: | |
| return handle.read() | |
| except OSError: | |
| return "" | |
| def _sysctl(name: str) -> str | None: | |
| try: | |
| import subprocess | |
| return subprocess.run(["sysctl", "-n", name], capture_output=True, | |
| text=True, timeout=5).stdout.strip() | |
| except Exception: | |
| return None | |
| def _powershell(command: str) -> str: | |
| import subprocess | |
| result = subprocess.run( | |
| ["powershell", "-NoProfile", "-NonInteractive", "-Command", command], | |
| capture_output=True, text=True, timeout=30) | |
| if result.returncode != 0: | |
| raise RuntimeError(result.stderr.strip() or "powershell failed") | |
| return result.stdout.strip() | |
| def _cache_linux() -> dict: | |
| caches: dict[int, int] = {} | |
| line_sizes: dict[int, int] = {} | |
| for index in range(16): | |
| base = f"/sys/devices/system/cpu/cpu0/cache/index{index}" | |
| text = _read_file(f"{base}/size").strip() | |
| if not text: | |
| continue | |
| level = int(_read_file(f"{base}/level").strip()) | |
| size_kb = int(text.rstrip("KkMm")) * (1024 if text.lower().endswith("m") else 1) | |
| cache_type = _read_file(f"{base}/type").strip() | |
| if cache_type in ("Data", "Unified") or level >= 3: | |
| caches[level] = max(caches.get(level, 0), size_kb) | |
| try: | |
| line_sizes[level] = int(_read_file(f"{base}/coherency_line_size").strip()) | |
| except (OSError, ValueError): | |
| pass | |
| return {"per_level_kb": {f"l{level}": size for level, size in sorted(caches.items())}, | |
| "line_sizes": {f"l{level}": size for level, size in sorted(line_sizes.items())}, | |
| "shared_l3_note": "L3 value is per-CCX/package as exposed by sysfs"} | |
| def _cache_windows() -> dict: | |
| import ctypes | |
| import struct | |
| kernel32 = ctypes.windll.kernel32 | |
| length = ctypes.c_ulong(0) | |
| kernel32.GetLogicalProcessorInformation(None, ctypes.byref(length)) | |
| if length.value == 0: | |
| raise RuntimeError("GetLogicalProcessorInformation returned empty length") | |
| buffer = (ctypes.c_char * length.value)() | |
| if not kernel32.GetLogicalProcessorInformation(buffer, ctypes.byref(length)): | |
| raise RuntimeError("GetLogicalProcessorInformation failed") | |
| count = length.value // 24 if platform.machine().endswith("64") else length.value // 24 | |
| # x64 entry layout: mask(8) + relationship(4) + padding(4) + union(16, | |
| # two ULONGLONGs) = 32 bytes; x86 uses 24. Never guess from the reported | |
| # length - derive stride from the platform ABI. | |
| stride = 32 if platform.machine().endswith("64") else 24 | |
| count = length.value // stride | |
| caches: dict[int, int] = {} | |
| line_sizes: dict[int, int] = {} | |
| for entry in range(count): | |
| offset = entry * stride | |
| relationship = struct.unpack_from("<I", buffer, offset + 8)[0] | |
| if relationship != 2: # RelationCache | |
| continue | |
| base = offset + 16 | |
| level = struct.unpack_from("<B", buffer, base)[0] | |
| if not 1 <= level <= 3: # Reject misaligned garbage instead of keying it. | |
| continue | |
| line_size = struct.unpack_from("<H", buffer, base + 2)[0] | |
| size_kb = struct.unpack_from("<I", buffer, base + 4)[0] // 1024 | |
| cache_type = struct.unpack_from("<i", buffer, base + 8)[0] | |
| # Type 1 = instruction, 2 = data, 0 = unified. | |
| if level >= 3 or cache_type in (0, 2): | |
| caches[level] = max(caches.get(level, 0), size_kb) | |
| line_sizes[level] = line_size | |
| return {"per_level_kb": {f"l{level}": size for level, size in sorted(caches.items())}, | |
| "line_sizes": {f"l{level}": size for level, size in sorted(line_sizes.items())}} | |
| def _cache_macos() -> dict: | |
| caches = {} | |
| for level, name in ((1, "hw.l1icachesize"), (2, "hw.l2cachesize"), (3, "hw.l3cachesize")): | |
| value = _sysctl(name) | |
| if value and value.isdigit(): | |
| caches[f"l{level}"] = int(value) // 1024 | |
| return {"per_level_kb": caches} | |
| def collect_isa() -> dict: | |
| # Prefer CISM's native CPUID probe: precise and vendor-agnostic (reports | |
| # VNNI/AVX-512 correctly on Windows, where the OS exposes almost nothing). | |
| try: | |
| import cism._native as native | |
| features = dict(native.cpu_features()) | |
| if features: | |
| features["neon"] = False | |
| features["asimddp"] = False | |
| return features | |
| except Exception: | |
| pass | |
| system = platform.system() | |
| flags: set[str] = set() | |
| if system == "Linux": | |
| for line in _read_file("/proc/cpuinfo").splitlines(): | |
| if line.lower().startswith("flags"): | |
| flags = set(line.split(":", 1)[1].lower().split()) | |
| break | |
| elif system == "Windows": | |
| import ctypes | |
| # PF_AVX2_INSTRUCTIONS_AVAILABLE = 40, PF_AVX512F_INSTRUCTIONS_AVAILABLE = 43. | |
| present = ctypes.windll.kernel32.IsProcessorFeaturePresent | |
| flags = {"avx2"} if present(40) else set() | |
| if present(43): | |
| flags.add("avx512f") | |
| flags.add("fma") if flags else flags # AVX2 CPUs in practice ship FMA. | |
| else: | |
| features = _sysctl("machdep.cpu.features") or "" | |
| flags = set(features.lower().split()) | |
| result = {name: (name in flags) for name in | |
| ("avx2", "fma", "avx512f", "avx512_vnni", "avx512_bf16")} | |
| # ARM rows (e.g. Apple platforms running x86-unaware Pythons) via os flags. | |
| result["neon"] = "neon" in flags or "asimd" in flags | |
| result["asimddp"] = "asimddp" in flags or "dotprod" in flags | |
| return result | |
| def collect_memory() -> dict: | |
| system = platform.system() | |
| if system == "Windows": | |
| import ctypes | |
| class MemoryStatus(ctypes.Structure): | |
| _fields_ = [("dwLength", ctypes.c_ulong), ("dwMemoryLoad", ctypes.c_ulong), | |
| ("ullTotalPhys", ctypes.c_ulonglong), ("ullAvailPhys", ctypes.c_ulonglong), | |
| ("ullTotalPageFile", ctypes.c_ulonglong), ("ullAvailPageFile", ctypes.c_ulonglong), | |
| ("ullTotalVirtual", ctypes.c_ulonglong), ("ullAvailVirtual", ctypes.c_ulonglong), | |
| ("ullAvailExtendedVirtual", ctypes.c_ulonglong)] | |
| status = MemoryStatus() | |
| status.dwLength = ctypes.sizeof(MemoryStatus) | |
| ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(status)) | |
| return {"total_gb": round(status.ullTotalPhys / 1e9, 2), | |
| "available_gb": round(status.ullAvailPhys / 1e9, 2)} | |
| if system == "Linux": | |
| total_kb, available_kb = 0, 0 | |
| for line in _read_file("/proc/meminfo").splitlines(): | |
| if line.startswith("MemTotal"): | |
| total_kb = int(line.split()[1]) | |
| if line.startswith("MemAvailable"): | |
| available_kb = int(line.split()[1]) | |
| return {"total_gb": round(total_kb / 1e6, 2), "available_gb": round(available_kb / 1e6, 2)} | |
| value = _sysctl("hw.memsize") | |
| return {"total_gb": round(int(value) / 1e9, 2)} if value and value.isdigit() else {} | |
| def collect_virtualization() -> dict: | |
| info: dict = {"detected": None, "vcpu_quota": None} | |
| system = platform.system() | |
| if system == "Linux": | |
| cpuinfo = _read_file("/proc/cpuinfo").lower() | |
| if "hypervisor" in cpuinfo or "vmware" in cpuinfo or "kvm" in cpuinfo: | |
| info["detected"] = "hypervisor flag in /proc/cpuinfo" | |
| cpu_max = _read_file("/sys/fs/cgroup/cpu.max").strip() | |
| if cpu_max and not cpu_max.startswith("max"): | |
| quota, period = cpu_max.split() | |
| info["vcpu_quota"] = round(int(quota) / int(period), 2) | |
| else: | |
| quota = _read_file("/sys/fs/cgroup/cpu/cpu.cfs_quota_us").strip() | |
| period = _read_file("/sys/fs/cgroup/cpu/cpu.cfs_period_us").strip() | |
| if quota and quota != "-1" and period: | |
| info["vcpu_quota"] = round(int(quota) / int(period), 2) | |
| elif system == "Windows": | |
| try: | |
| result = _powershell( | |
| "Get-CimInstance Win32_ComputerSystem | Select-Object -First 1 " | |
| "Manufacturer,Model | ConvertTo-Json -Compress") | |
| import json as _json | |
| data = _json.loads(result) | |
| identity = f"{data.get('Manufacturer', '')} {data.get('Model', '')}" | |
| info["system_identity"] = identity.strip() | |
| markers = ("virtual", "vmware", "kvm", "qemu", "xen", "hyper-v", "bochs") | |
| if any(marker in identity.lower() for marker in markers): | |
| info["detected"] = identity.strip() | |
| except Exception: | |
| pass | |
| return info | |
| def probe_bandwidth(size_mb: float, reps: int = 5) -> float: | |
| """Single-core sequential read bandwidth via numpy count_nonzero, in GB/s. | |
| uint8 count_nonzero is a pure streaming read; float32 arithmetic reductions | |
| are compute-limited and undercount real bandwidth about 5x. | |
| """ | |
| import numpy as np | |
| buffer = np.zeros(int(size_mb * 1e6), dtype=np.uint8) | |
| buffer[::4096] = 1 # Defeat trivially-skippable all-zero fast paths. | |
| best = float("inf") | |
| for _ in range(reps): | |
| start = time.perf_counter() | |
| found = np.count_nonzero(buffer) | |
| elapsed = time.perf_counter() - start | |
| best = min(best, elapsed) | |
| if found == buffer.size: | |
| raise RuntimeError("impossible count") | |
| return buffer.nbytes / best / 1e9 | |
| def probe_thread_sync(rounds: int = 20000) -> float: | |
| """Ping-pong between two threads; returns microseconds per round trip.""" | |
| ping = threading.Event() | |
| pong = threading.Event() | |
| ping.set() | |
| def responder(): | |
| for _ in range(rounds): | |
| ping.wait() | |
| ping.clear() | |
| pong.set() | |
| worker = threading.Thread(target=responder, daemon=True) | |
| worker.start() | |
| start = time.perf_counter() | |
| for _ in range(rounds): | |
| pong.wait() | |
| pong.clear() | |
| ping.set() | |
| worker.join(timeout=5) | |
| elapsed = time.perf_counter() - start | |
| return elapsed / rounds * 1e6 | |
| def cism_probe() -> dict | None: | |
| try: | |
| import cism | |
| import cism._native as native | |
| except ImportError: | |
| return None | |
| info: dict = {"version": cism.__version__} | |
| # Smoke check with a tiny in-memory model; not a performance claim. | |
| config = {"model_type": "llama", "hidden_size": 24, "intermediate_size": 37, | |
| "num_hidden_layers": 2, "num_attention_heads": 4, "num_key_value_heads": 2, | |
| "head_dim": 8, "vocab_size": 43, "max_position_embeddings": 80, | |
| "rms_norm_eps": 1e-5, "rope_theta": 10000.0, "tie_word_embeddings": True} | |
| try: | |
| import numpy as np | |
| rng = np.random.default_rng(7) | |
| weights = {"model.embed_tokens.weight": rng.normal(0, 0.1, (43, 24)).astype(np.float32), | |
| "model.norm.weight": np.ones(24, dtype=np.float32)} | |
| for layer in range(2): | |
| prefix = f"model.layers.{layer}." | |
| weights[prefix + "input_layernorm.weight"] = np.ones(24, dtype=np.float32) | |
| weights[prefix + "post_attention_layernorm.weight"] = np.ones(24, dtype=np.float32) | |
| weights[prefix + "self_attn.q_proj.weight"] = rng.normal(0, 0.1, (32, 24)).astype(np.float32) | |
| weights[prefix + "self_attn.k_proj.weight"] = rng.normal(0, 0.1, (16, 24)).astype(np.float32) | |
| weights[prefix + "self_attn.v_proj.weight"] = rng.normal(0, 0.1, (16, 24)).astype(np.float32) | |
| weights[prefix + "self_attn.o_proj.weight"] = rng.normal(0, 0.1, (24, 32)).astype(np.float32) | |
| weights[prefix + "mlp.gate_proj.weight"] = rng.normal(0, 0.1, (37, 24)).astype(np.float32) | |
| weights[prefix + "mlp.up_proj.weight"] = rng.normal(0, 0.1, (37, 24)).astype(np.float32) | |
| weights[prefix + "mlp.down_proj.weight"] = rng.normal(0, 0.1, (24, 37)).astype(np.float32) | |
| model = native.Model(config, weights, "fp32") | |
| session = model.create_session([1, 2, 3], max_new_tokens=2, temperature=0, | |
| top_p=1, top_k=0, seed=0, eos_token_ids=[]) | |
| tokens = session.next_tokens(2) | |
| info["native_smoke_test"] = "pass" if len(tokens) == 2 else "no tokens" | |
| info["kernel"] = model.info["kernel"] | |
| info["threads"] = model.info["threads"] | |
| info["precision"] = model.info["precision"] | |
| except Exception as error: | |
| info["native_smoke_test"] = f"fail: {error}" | |
| return info | |
| def verdict(data: dict) -> list[str]: | |
| lines = [] | |
| cache = (data.get("cpu") or {}).get("cache") or {} | |
| per_level = cache.get("per_level_kb", {}) | |
| l3_kb = per_level.get("l3") | |
| if l3_kb: | |
| l3_mb = l3_kb / 1024 | |
| lines.append(f"L3 cache: {l3_mb:.1f} MB -> CISM cache-resident ceiling is about " | |
| f"{l3_mb * 0.8:.0f}-{l3_mb * 0.9:.0f} MB of packed weights " | |
| "(leaving headroom for KV, activations, and the OS).") | |
| ladder = data.get("bandwidth_ladder", {}) | |
| rates = [value.get("gbps") for value in ladder.values() if value.get("gbps")] | |
| if len(rates) >= 3: | |
| small, large = rates[0], rates[-1] | |
| if large > 0: | |
| lines.append(f"Measured bandwidth ladder: {rates[0]:.1f} -> {rates[1]:.1f} -> " | |
| f"{rates[2]:.1f} GB/s (small -> mid -> large buffer). " | |
| f"Cache-to-RAM multiplier on this machine: about {small / large:.1f}x.") | |
| sync = data.get("thread_sync_us_per_round") | |
| if sync is not None: | |
| lines.append(f"Thread ping-pong: {sync:.2f} us/round trip. Sub-microsecond models " | |
| f"(<= 1 MB) will lose most of their time to barriers like this; " | |
| f"single-thread them.") | |
| isa = data.get("isa", {}) | |
| if isa.get("avx512_vnni"): | |
| lines.append("AVX-512 VNNI available: integer-int8 kernels can use hardware " | |
| "dot products here.") | |
| elif isa.get("avx2"): | |
| lines.append("AVX2 available: CISM uses the FMA path (no hardware VNNI; " | |
| "expect compute-bound int8 kernels).") | |
| quota = (data.get("virtualization") or {}).get("vcpu_quota") | |
| if quota is not None: | |
| lines.append(f"cgroup vCPU quota: {quota:.2f} CPUs - multithreaded decode beyond " | |
| f"this quota will be throttled.") | |
| return lines | |
| def main() -> int: | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--json", action="store_true", help="emit JSON only") | |
| parser.add_argument("--fast", action="store_true", help="smaller probe buffers") | |
| parser.add_argument("--skip-bench", action="store_true", help="skip measured probes") | |
| args = parser.parse_args() | |
| report: dict = {"platform": collect_platform(), "cpu": collect_cpu_topology(), | |
| "isa": collect_isa(), "memory": collect_memory(), | |
| "virtualization": collect_virtualization()} | |
| if not args.skip_bench: | |
| try: | |
| import numpy # noqa: F401 | |
| except ImportError: | |
| report["bandwidth_ladder"] = {"error": "numpy is required for probes"} | |
| else: | |
| rungs = ((0.5, "l2_class"), (16, "l3_class"), (256, "dram_class")) if not args.fast \ | |
| else ((0.5, "l2_class"), (4, "l3_class"), (64, "dram_class")) | |
| ladder = {} | |
| for size_mb, name in rungs: | |
| try: | |
| ladder[name] = {"buffer_mb": size_mb, | |
| "gbps": round(probe_bandwidth(size_mb), 2)} | |
| except Exception as error: | |
| ladder[name] = {"error": str(error)} | |
| report["bandwidth_ladder"] = ladder | |
| report["thread_sync_us_per_round"] = round(probe_thread_sync(), 3) | |
| report["cism"] = cism_probe() | |
| report["interpretation"] = verdict(report) | |
| if args.json: | |
| print(json.dumps(report, indent=2)) | |
| return 0 | |
| print(json.dumps(report, indent=2)) | |
| print("\n" + "=" * 78) | |
| print("INTERPRETATION") | |
| print("=" * 78) | |
| for line in report["interpretation"]: | |
| print(f"- {line}") | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |