File size: 2,559 Bytes
95200a7 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 | """What this machine would put on its GPU, and why that much.
Printed at the start of a benchmark or a support conversation. Every number
here is one the worker actually uses, read the way the worker reads it, so a
disagreement between this and what happened is a bug rather than a mystery.
"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from distinct_agent import runtime # noqa: E402
from distinct_agent.gguf import read_shape # noqa: E402
from distinct_agent.server_runner import ( # noqa: E402
VRAM_BUDGET_SHARE,
VRAM_FIXED_OVERHEAD,
plan_offload,
)
from distinct_agent.weights import ( # noqa: E402
DEFAULT_CACHE_GB,
WeightsCache,
cache_bytes,
fetchable_manifests,
shared_cache_directory,
)
MB = 1024 * 1024
def main() -> int:
free, total = runtime.accelerator_memory()
print(f"platform {runtime.platform_key()}")
print(f"cuda ceiling {runtime.cuda_capability()}")
print(f"vram {free / MB:.0f} MB free of {total / MB:.0f} MB")
print(f"budget share {VRAM_BUDGET_SHARE} minus {VRAM_FIXED_OVERHEAD / MB:.0f} MB overhead")
print(f"preferred {', '.join(runtime.preferred_keys())}")
print(f"installed {runtime.best_installed() or 'nothing'}")
cache = WeightsCache(
shared_cache_directory(),
limit_bytes=cache_bytes(DEFAULT_CACHE_GB),
manifests=tuple(fetchable_manifests()),
)
found = list(cache.installed())
if not found:
print("models none cached")
return 0
for model in found:
size = model.path.stat().st_size
shape = read_shape(model.path)
context = min(4096, model.manifest.context_length)
plan = plan_offload(
model_bytes=size,
shape=shape,
free_vram_bytes=free,
context_tokens=context,
)
per_layer = (
size / (shape.layers + 1) + shape.kv_bytes_per_layer(context)
if shape.known
else 0
)
print(
f"model {model.manifest.id} ({size / MB:.0f} MB)\n"
f" layers {shape.layers}, {shape.heads} heads, "
f"{shape.kv_heads} kv heads, dim {shape.head_dimension}\n"
f" per layer {per_layer / MB:.0f} MB "
f"(weights + kv at {context} tokens)\n"
f" offload plan {plan}"
)
return 0
if __name__ == "__main__":
raise SystemExit(main())
|