distinct / scripts /offload_report.py
User1342's picture
Offload what the card can hold, survive a dropped connection, and stop sending prompts the model cannot read
95200a7
Raw History Blame Contribute Delete
2.56 kB
"""What this machine would put on its GPU, and why that much.
Printed at the start of a benchmark or a support conversation. Every number
here is one the worker actually uses, read the way the worker reads it, so a
disagreement between this and what happened is a bug rather than a mystery.
"""
from __future__ import annotations
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from distinct_agent import runtime # noqa: E402
from distinct_agent.gguf import read_shape # noqa: E402
from distinct_agent.server_runner import ( # noqa: E402
VRAM_BUDGET_SHARE,
VRAM_FIXED_OVERHEAD,
plan_offload,
)
from distinct_agent.weights import ( # noqa: E402
DEFAULT_CACHE_GB,
WeightsCache,
cache_bytes,
fetchable_manifests,
shared_cache_directory,
)
MB = 1024 * 1024
def main() -> int:
free, total = runtime.accelerator_memory()
print(f"platform {runtime.platform_key()}")
print(f"cuda ceiling {runtime.cuda_capability()}")
print(f"vram {free / MB:.0f} MB free of {total / MB:.0f} MB")
print(f"budget share {VRAM_BUDGET_SHARE} minus {VRAM_FIXED_OVERHEAD / MB:.0f} MB overhead")
print(f"preferred {', '.join(runtime.preferred_keys())}")
print(f"installed {runtime.best_installed() or 'nothing'}")
cache = WeightsCache(
shared_cache_directory(),
limit_bytes=cache_bytes(DEFAULT_CACHE_GB),
manifests=tuple(fetchable_manifests()),
)
found = list(cache.installed())
if not found:
print("models none cached")
return 0
for model in found:
size = model.path.stat().st_size
shape = read_shape(model.path)
context = min(4096, model.manifest.context_length)
plan = plan_offload(
model_bytes=size,
shape=shape,
free_vram_bytes=free,
context_tokens=context,
)
per_layer = (
size / (shape.layers + 1) + shape.kv_bytes_per_layer(context)
if shape.known
else 0
)
print(
f"model {model.manifest.id} ({size / MB:.0f} MB)\n"
f" layers {shape.layers}, {shape.heads} heads, "
f"{shape.kv_heads} kv heads, dim {shape.head_dimension}\n"
f" per layer {per_layer / MB:.0f} MB "
f"(weights + kv at {context} tokens)\n"
f" offload plan {plan}"
)
return 0
if __name__ == "__main__":
raise SystemExit(main())