Download scripts/offload_report.py from User1342/distinct: direct link, hf CLI and curl.
- Browser
- Download file 2.56 kB
-
https://huggingface.co/spaces/User1342/distinct/resolve/main/scripts/offload_report.py
- Command line
-
hf download hf://spaces/User1342/distinct/scripts/offload_report.py
-
curl -L -o offload_report.py https://huggingface.co/spaces/User1342/distinct/resolve/main/scripts/offload_report.py
2.56 kB
| """What this machine would put on its GPU, and why that much. | |
| Printed at the start of a benchmark or a support conversation. Every number | |
| here is one the worker actually uses, read the way the worker reads it, so a | |
| disagreement between this and what happened is a bug rather than a mystery. | |
| """ | |
| from __future__ import annotations | |
| import sys | |
| from pathlib import Path | |
| sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) | |
| from distinct_agent import runtime # noqa: E402 | |
| from distinct_agent.gguf import read_shape # noqa: E402 | |
| from distinct_agent.server_runner import ( # noqa: E402 | |
| VRAM_BUDGET_SHARE, | |
| VRAM_FIXED_OVERHEAD, | |
| plan_offload, | |
| ) | |
| from distinct_agent.weights import ( # noqa: E402 | |
| DEFAULT_CACHE_GB, | |
| WeightsCache, | |
| cache_bytes, | |
| fetchable_manifests, | |
| shared_cache_directory, | |
| ) | |
| MB = 1024 * 1024 | |
| def main() -> int: | |
| free, total = runtime.accelerator_memory() | |
| print(f"platform {runtime.platform_key()}") | |
| print(f"cuda ceiling {runtime.cuda_capability()}") | |
| print(f"vram {free / MB:.0f} MB free of {total / MB:.0f} MB") | |
| print(f"budget share {VRAM_BUDGET_SHARE} minus {VRAM_FIXED_OVERHEAD / MB:.0f} MB overhead") | |
| print(f"preferred {', '.join(runtime.preferred_keys())}") | |
| print(f"installed {runtime.best_installed() or 'nothing'}") | |
| cache = WeightsCache( | |
| shared_cache_directory(), | |
| limit_bytes=cache_bytes(DEFAULT_CACHE_GB), | |
| manifests=tuple(fetchable_manifests()), | |
| ) | |
| found = list(cache.installed()) | |
| if not found: | |
| print("models none cached") | |
| return 0 | |
| for model in found: | |
| size = model.path.stat().st_size | |
| shape = read_shape(model.path) | |
| context = min(4096, model.manifest.context_length) | |
| plan = plan_offload( | |
| model_bytes=size, | |
| shape=shape, | |
| free_vram_bytes=free, | |
| context_tokens=context, | |
| ) | |
| per_layer = ( | |
| size / (shape.layers + 1) + shape.kv_bytes_per_layer(context) | |
| if shape.known | |
| else 0 | |
| ) | |
| print( | |
| f"model {model.manifest.id} ({size / MB:.0f} MB)\n" | |
| f" layers {shape.layers}, {shape.heads} heads, " | |
| f"{shape.kv_heads} kv heads, dim {shape.head_dimension}\n" | |
| f" per layer {per_layer / MB:.0f} MB " | |
| f"(weights + kv at {context} tokens)\n" | |
| f" offload plan {plan}" | |
| ) | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |