File size: 2,559 Bytes
95200a7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
"""What this machine would put on its GPU, and why that much.

Printed at the start of a benchmark or a support conversation. Every number
here is one the worker actually uses, read the way the worker reads it, so a
disagreement between this and what happened is a bug rather than a mystery.
"""

from __future__ import annotations

import sys
from pathlib import Path

sys.path.insert(0, str(Path(__file__).resolve().parent.parent))

from distinct_agent import runtime  # noqa: E402
from distinct_agent.gguf import read_shape  # noqa: E402
from distinct_agent.server_runner import (  # noqa: E402
    VRAM_BUDGET_SHARE,
    VRAM_FIXED_OVERHEAD,
    plan_offload,
)
from distinct_agent.weights import (  # noqa: E402
    DEFAULT_CACHE_GB,
    WeightsCache,
    cache_bytes,
    fetchable_manifests,
    shared_cache_directory,
)

MB = 1024 * 1024


def main() -> int:
    free, total = runtime.accelerator_memory()
    print(f"platform        {runtime.platform_key()}")
    print(f"cuda ceiling    {runtime.cuda_capability()}")
    print(f"vram            {free / MB:.0f} MB free of {total / MB:.0f} MB")
    print(f"budget share    {VRAM_BUDGET_SHARE} minus {VRAM_FIXED_OVERHEAD / MB:.0f} MB overhead")
    print(f"preferred       {', '.join(runtime.preferred_keys())}")
    print(f"installed       {runtime.best_installed() or 'nothing'}")

    cache = WeightsCache(
        shared_cache_directory(),
        limit_bytes=cache_bytes(DEFAULT_CACHE_GB),
        manifests=tuple(fetchable_manifests()),
    )
    found = list(cache.installed())
    if not found:
        print("models          none cached")
        return 0
    for model in found:
        size = model.path.stat().st_size
        shape = read_shape(model.path)
        context = min(4096, model.manifest.context_length)
        plan = plan_offload(
            model_bytes=size,
            shape=shape,
            free_vram_bytes=free,
            context_tokens=context,
        )
        per_layer = (
            size / (shape.layers + 1) + shape.kv_bytes_per_layer(context)
            if shape.known
            else 0
        )
        print(
            f"model           {model.manifest.id} ({size / MB:.0f} MB)\n"
            f"  layers        {shape.layers}, {shape.heads} heads, "
            f"{shape.kv_heads} kv heads, dim {shape.head_dimension}\n"
            f"  per layer     {per_layer / MB:.0f} MB "
            f"(weights + kv at {context} tokens)\n"
            f"  offload plan  {plan}"
        )
    return 0


if __name__ == "__main__":
    raise SystemExit(main())