Winery-Strata / setup.py
Zandy-Wandy's picture
🍷 Winery 9B cellar: the dense 9B blends at Q8_0 on llama.cpp (fork commit a75a246)
31b6789 verified
Raw History Blame Contribute Delete
227 kB
#!/usr/bin/env python3
"""Strata one-click setup and start (Windows and Linux, NVIDIA or AMD GPUs).
START-HERE.bat (Windows) / ./setup.sh (Linux) - they install Python if needed and run this file
The first time it asks four questions - which model (the original Qwen3.8-Flash-Next or the Swift 1.5 fine-tune),
which size, how much context, and whether the model should also read images - then installs everything and starts the model on http://127.0.0.1:8080 (OpenAI- and Anthropic-compatible
API; a small page there shows that it runs). Every later start skips straight to running the model: nothing that
is already downloaded, installed or prepared is done again.
What the first run does (each step is skipped when it is already done):
1. checks your PC: NVIDIA or AMD GPU and driver, RAM, CPU, free disk space
2. asks the questions
3. installs the Python packages it needs into .venv (numpy, jinja2, ..., and NVIDIA's CUDA libraries)
4. gets the Strata engine: a ready-made build for RTX 20/30/40/50 cards (no compiler needed); if none fits your PC,
it installs the build tools (asks first) and compiles the engine for your GPU. AMD (--backend hip, chosen by
itself on a PC with no usable NVIDIA card): the ready-made HIP engine on Windows, compiled here on Linux
5. downloads the model from Hugging Face (resumable), and the vision encoder if you want images
6. prepares the model for Strata and fetches the MTP draft layer (~5 GB, from the original Qwen checkpoint)
7. writes run-<model>.bat / run-<model>.sh and starts the model
Options: --family winery|qwen|swift|coder|unsloth, --model SLIM|RESERVE|GRAND (winery), --model Q2_0|IQ2_XS|IQ3_XXS|IQ3_S, --context 32768, --rope-scaling none|linear|yarn
(--rope-scale F; past the trained 262144 the setup adds yarn and the factor is the final context over 262144,
at least 1 - an explicit --rope-scaling none is refused for such a context), --vision yes|no|gpu|cpu, --port
8080, --yes (recommended
answers, no questions), --setup (install another model / change settings instead of starting), --no-start,
--host 0.0.0.0 --api-key KEY (reach it from other devices on your network), --experimental-speed-projection on|off
(EXPERIMENTAL, off by default),
--models-dir DIR, --gguf-dir DIR (use GGUF files you already have), --build (compile instead of the ready-made
engine), --check (only check this PC), --resident-budget-gib N (UD-Q4_K_XL's experts in RAM), --kv-streaming
on|off|auto.
Setup recommends, it never forces: the recommended answers are the defaults (--yes, or Enter), and a bigger choice
than it recommends - a longer context, more GPUs, a bigger RAM budget, a size it thinks will not fit - is kept, with
what it risks. With --yes, an explicit flag (--model, --gpus, ...) is the consent to a risk setup would otherwise
stop at; --yes alone is not.
"""
from __future__ import annotations
import argparse
import ctypes
import hashlib
import json
import math
import os
import platform
import re
import shutil
import struct
import subprocess
import sys
import textwrap
import time
import urllib.error
import urllib.request
import zipfile
from pathlib import Path
ROOT = Path(__file__).resolve().parent
WIN = os.name == "nt"
sys.path.insert(0, str(ROOT))
import cellar # noqa: E402 🍷 the Winery 9B blends on llama.cpp
# #214: every Hugging Face file comes from a fixed commit of its repository (the `sha` of
# https://huggingface.co/api/models/<repo> when this was pinned), so a checkout installs the same files on any
# day. A revision the repository no longer has falls back to its current files, with a message (download()).
HF_REVISIONS = {
"ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF": "ed59f92082b1e93c0e96d60a8b11aab089b52f09", # 2026-09-29
"ukisai/Swift-1.5-Qwen3.8-Flash-Next-GSQ-RCO-GGUF": "b22d729eae29b5796f76fb70f91aef549b9fc52c", # 2026-09-24
"ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-Coder-GGUF": "5348543e0147355ac9cbcb031184a3546350988e", # 2026-09-29
"unsloth/Qwen3.8-Flash-Next-GGUF": "38bb39ee97821de2c9009abb7e93950eec396e66", # 2026-09-30
# 🍷 Winery fork: WineryLabs' Flash-Next blend (pinned after each upload; tools/winery_pin.py rewrites it)
"WineryLabs/Winery-Qwen3.8-Flash-Next-Cuvee-GGUF": "main",
}
HF_DEFAULT = "https://huggingface.co"
def hf_endpoint() -> str:
"""#495: the Hugging Face host - HF_ENDPOINT as huggingface_hub reads it (a mirror, e.g. https://hf-mirror.com),
else huggingface.co. The pinned revisions and the SHA-256 checks are the same whichever host serves the files."""
return (os.environ.get("HF_ENDPOINT") or "").strip().rstrip("/") or HF_DEFAULT
def hf(repo: str) -> str:
"""The download folder of a Hugging Face repository at its pinned revision."""
return f"{hf_endpoint()}/{repo}/resolve/{HF_REVISIONS[repo]}/"
def hf_unpinned(url: str) -> str:
"""The same file at the repository's current revision (main)."""
return re.sub(r"^(https?://[^/]+/.+?/resolve/)[0-9a-f]{40}/", r"\1main/", url, count=1)
HF = hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF")
LLAMA_CPP_COMMIT = "3cf03257f219afbe7334045ff7c6a06ac68c627d"
LLAMA_CPP_ZIP = f"https://github.com/ggml-org/llama.cpp/archive/{LLAMA_CPP_COMMIT}.zip"
# The ready-made engine: <PREBUILT_URL><asset>, a zip with strata(.exe), strata-vision(.exe) and BUILD.json, built
# by tools/make_release.py. Set this to the GitHub release download folder when publishing, e.g.
# "https://github.com/<you>/Strata/releases/latest/download/" (or pass --prebuilt / set STRATA_PREBUILT_URL).
# With the default, the release of this checkout's own version (PREBUILT_TAG_URL, CMakeLists.txt's version) is
# tried first and the latest release is the fallback (#214): an older checkout keeps the engine it shipped with.
PREBUILT_URL = "https://github.com/Niko1221/Strata/releases/latest/download/"
PREBUILT_TAG_URL = "https://github.com/Niko1221/Strata/releases/download/v{version}/"
PREBUILT_ASSET = "strata-windows-x64.zip" if WIN else "strata-linux-x64.zip"
# the CUDA libraries the ready-made engine loads (the same CUDA 13.0 it is built with), from NVIDIA's pip packages
CUDA_WHEELS = ["nvidia-cublas==13.0.2.14", "nvidia-cuda-runtime==13.0.96"]
MIN_DRIVER = 580 # CUDA 13.0
MIN_ENGINE = (0, 1, 38) # v0.1.38: prompts faster (one gather per expert group #372, the first chunk's PLE rows beside layer 0 #374, DeltaNet three heads per thread #413), --kv q4_0 prompts on tensor cores (#452), Q5_0 experts on the GPU (#473), IQ4_XS on AVX-2 (#415), unbuffered expert loading on Windows (#357 #362), --peer-device (#531), a 6 GB card starts (#496), PR batch; v0.1.37: a silent engine is restarted (#481), Windows AMD counts the desktop's VRAM (#380 #377 #497), a steadier PCIe probe (#485), fixes #496 #495 #498 #505 #493; v0.1.36: a cancelled prompt logged as read so far (#471), the draft-head hint (#474), UPDATE.bat (#475), --expert-profile-save (#477); v0.1.35: Windows AMD uses its bundled HIP runtime (#468 #461), the low-RAM resident mode on Windows 32 GB (#467), fixes #460 #459 #446 #447 #457 #448 #444; v0.1.34: AMD on Windows (a ready-made HIP engine), an MCP server for AI assistants (tools/strata_mcp.py), a shorter README; v0.1.33: a portable image encoder again (#411 #412), setup recommends instead of forcing (#406 #403 #364 #384), fixes #352 #365 #369 #371 #375 #393 #408 #414; v0.1.32: split prompts faster (#340), AMD router +12%, Unsloth Q4 in setup, faster Q4 prompts, #326/#327/#342/#344 fixes, PR batch; v0.1.31: Unsloth UD-Q4_K_XL (experimental), GGUF-in-place low-RAM mode, Windows GGUF load 2x, server race + tokenizer fixes, AMD intrinsics; v0.1.30: short prompts faster (streaming from 1024 tokens), resident low-RAM variant, multi-GPU session carve, RDNA4; v0.1.29: sampled answers faster (split top-k), #154 correctness fixes; v0.1.28: the expert cache reserves the draft head, a cancelled request no longer fails the next; v0.1.27: RTX 20 (sm_75) in the ready-made engine, the HIP build without CUDA headers; v0.1.26: the draft layer's prompt pass in batches; v0.1.25: faster prompts (grouping off the copy engine, fused hyper-connection kernels), AMD HIP backend, --kv k8v4; v0.1.24: long prompts faster (QSA select on tensor cores); v0.1.23: image requests honor sampling, 8 GB cards start, batched verify window; v0.1.22: faster prompts (tensor-core attention), multi-GPU across images/steering/KV streaming; v0.1.21: multi-GPU layer split (--gpus); v0.1.20: system-prompt checkpoint, PCIe probe, hit rate; v0.1.19: penalties
PY_PACKAGES = ["numpy", "jinja2", "regex", "pyyaml", "tqdm", "requests", "cmake", "ninja", "pillow", "psutil"]
REQUIREMENTS = ROOT / "requirements.txt" # the same packages and their dependencies, pinned (#214)
MODELS = {
# the original model only for now: Swift 1.5's Q2_0 files split one layer's experts across the two shards, which
# the pack tool (tools/iq_pack.py) cannot prepare yet (#171)
"Q2_0": {"about": "2-bit, the fastest", "download_gb": 66.4, "ram_gb": 48, "arena_gb": 34.0, "families": ("qwen",)},
"IQ2_XS": {"about": "2-bit i-quant, a little better quality, close in speed", "download_gb": 68.0, "ram_gb": 48,
"arena_gb": 35.5},
"IQ3_XXS": {"about": "3-bit i-quant, better quality, slower (more CPU work per token)", "download_gb": 75.8,
"ram_gb": 60, "arena_gb": 42.9},
# the original model only (Swift 1.5 has no IQ3_S): matches the full BF16 model on the published benchmarks
"IQ3_S": {"about": "3.5-bit i-quant, the best quality (matches the full model), the slowest; needs a 64 GB PC "
"with little else running", "download_gb": 83.6, "ram_gb": 62, "arena_gb": 50.3,
"families": ("qwen",)},
# the Coder release: 256 of the 512 experts kept (the ones code, tools and vision use), IQ2_S-IQ4_XS like IQ3_S
"IQ1_M": {"about": "the Coder's only size: half the experts, stored like IQ3_S (3.5 bits)", "download_gb": 58.4,
"ram_gb": 32, "arena_gb": 23.4, "families": ("coder",)},
# EXPERIMENTAL (docs/UNSLOTH_Q4.md): Unsloth's 4-bit file; its 77 GB of experts do not fit a 64 GB PC, so the engine
# keeps a RAM budget of them (--resident-budget-gib, chosen below) and reads the rest from the GGUF on the SSD
"UD-Q4_K_XL": {"about": "4-bit (Unsloth Dynamic), EXPERIMENTAL: the best quality, but most experts come from the "
"SSD on a 64 GB PC (7-8.5 tokens/s measured)", "download_gb": 111.3, "ram_gb": 48,
"arena_gb": 77.0, "families": ("unsloth",), "budget": True},
# 🍷 Winery Cuvée tiers (family "winery"): one blend, sized by RAM. arena_gb = the experts' bytes.
# POCKET: 8 GB cards (RTX 4060/4070 laptops, 3060 Ti...) - Slim's 256 experts in 2-bit and upstream Q2_0's lean
# dense layout (Q4_K attention, IQ4_XS/Q5_0 shared expert, Q5_K output, Q3_K embeddings), so the dense weights
# leave an 8 GB card room for an expert cache; 16 GB laptops run it in the low-RAM mode
"POCKET": {"about": "Winery Pocket: Slim's 256 experts in 2-bit (IQ2_XS/Q2_0) and lean dense weights; for 8 GB "
"GPUs and 16-32 GB laptops", "download_gb": 48.0, "ram_gb": 24, "arena_gb": 17.3,
"families": ("winery",), "shards": 2, "profile": "expert-profile-winery-slim.bin"},
"SLIM": {"about": "Winery Slim: the 256 most-routed experts of each layer (picked on general text, not code), "
"IQ3_XXS/IQ4_NL; for 32 GB PCs", "download_gb": 60.0, "ram_gb": 32, "arena_gb": 26.7,
"families": ("winery",), "shards": 2, "profile": "expert-profile-winery-slim.bin"},
"RESERVE": {"about": "Winery Reserve: all 512 experts, IQ3_XXS/IQ4_NL (the Orca-validated layout); for 64 GB PCs",
"download_gb": 86.0, "ram_gb": 64, "arena_gb": 53.5, "families": ("winery",), "shards": 2},
"GRAND": {"about": "Winery Grand Cru: all 512 experts in 4-bit (Q4_K/Q5_1, Unsloth's layout), the closest to "
"the full blend; all in RAM from 96-128 GB, partly from the SSD below that",
"download_gb": 113.0, "ram_gb": 96, "arena_gb": 76.0, "families": ("winery",), "shards": 3,
"budget": True},
}
# The experimental Unsloth file's four shards at the pinned revision: name -> (bytes, sha256), checked after the
# download (setup trusts no other model file by name and size alone either: check_shards reads their directories).
UNSLOTH_SHARDS = {
"Qwen3.8-Flash-Next-UD-Q4_K_XL-00001-of-00004.gguf":
(10946624, "4448186216b3af4cc558bbce2c3213f01608f8f8b2e5267a9767971dd3ec8082"),
"Qwen3.8-Flash-Next-UD-Q4_K_XL-00002-of-00004.gguf":
(49859583136, "3f342f1c1580473f1ee94ddd5b28206e8c07a70fa1a366f59d1d6c922919a6c9"),
"Qwen3.8-Flash-Next-UD-Q4_K_XL-00003-of-00004.gguf":
(49376141504, "56758f40269cad5cd9b0d3d6fbae0f40f6d5be6de49e4ab392dbe83157d9cbd3"),
"Qwen3.8-Flash-Next-UD-Q4_K_XL-00004-of-00004.gguf":
(12087983520, "753bda48b98ba4f1636134a90a967de1b2d3908a236c026e464777342e53510a"),
}
UNSLOTH_ENGINE = (0, 1, 32) # the first engine setup configures for UD-Q4_K_XL (0.1.31 ran it by hand)
UNSLOTH_RAM_LEFT_GB = 24 # RAM beside the budget: the OS, the engine, and the file cache the rest is read through
# Contexts past 262144 (the model's trained length) extend it by rope scaling: for the context it will
# serve the setup resolves the method (yarn, or one question when interactive) and derives the factor
# from the final context (final / 262144, at least 1) itself (below), keeps an explicit
# --rope-scaling/--rope-scale, and refuses an explicit --rope-scaling none there - the stock angles past
# the trained range are out of spec.
CONTEXTS = [8192, 32768, 65536, 131072, 262144, 393216, 524288]
# The model families: the same architecture, weights in the same three GSQ-RCO sizes, different files.
FAMILIES = {
# 🍷 the Winery fork's own model: Swift 1.5's short thinking + Huihui's abliteration, merged by task arithmetic
"winery": {"title": "Winery Cuvée", "by": "WineryLabs' blend of Qwen3.8-Flash-Next fine-tunes",
"about": "Swift 1.5's short thinking + Huihui's abliteration in one model; Pocket (8 GB GPUs), "
"Slim (32 GB), Reserve (64 GB) or Grand Cru (96-128 GB+)",
"hf": hf("WineryLabs/Winery-Qwen3.8-Flash-Next-Cuvee-GGUF") + "{q}/",
"file": "Winery-Cuvee-{q}-0000{i}-of-0000{n}.gguf", "tag": "winery-",
"mmproj_hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF"),
"mmproj": "mmproj-Qwen3.8-Flash-Next-BF16.gguf", "name": "winery-cuvee",
"pack_args": ["--compat-bf16"],
"license": "Qwen license (base) + Swift Open License 1.0 (Swift 1.5 weights): "
"https://huggingface.co/WineryLabs/Winery-Qwen3.8-Flash-Next-Cuvee-GGUF"},
"qwen": {"title": "Qwen3.8-Flash-Next", "by": "Qwen; GSQ-RCO quants by ISTA-DASLab",
"about": "the original model",
"hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF") + "{q}/",
"file": "Qwen3.8-Flash-Next-GSQ-RCO-{q}-0000{i}-of-00002.gguf", "tag": "",
"mmproj_hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF"),
"mmproj": "mmproj-Qwen3.8-Flash-Next-BF16.gguf", "name": "qwen3.8-flash-next"},
"swift": {"title": "Swift 1.5", "by": "UkisAI's fine-tune of Qwen3.8-Flash-Next",
"about": "thinks much shorter (-63% thinking tokens, 1.8x sooner answers by its authors' numbers)",
"hf": hf("ukisai/Swift-1.5-Qwen3.8-Flash-Next-GSQ-RCO-GGUF"),
"file": "Swift-Qwen3.8-Flash-Next-GSQ-RCO-{q}-0000{i}-of-00002.gguf", "tag": "swift-",
"mmproj_hf": hf("ukisai/Swift-1.5-Qwen3.8-Flash-Next-GSQ-RCO-GGUF"),
"mmproj": "mmproj-Swift-Qwen3.8-Flash-Next-BF16.gguf", "name": "swift-1.5",
"license": "Swift Open License 1.0: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-Flash-Next-GSQ-RCO-GGUF"},
# ISTA-DASLab's expert-pruned release: half of each layer's experts removed, chosen for code, agentic tool use and
# vision; its shard 2 (the n-gram table) and vision encoder are the original's files, shared with it
"coder": {"title": "Qwen3.8-Flash-Next Coder", "by": "ISTA-DASLab's coding version",
"about": "half the experts (code, tools, images kept): needs ~32 GB of RAM, faster; weaker outside coding",
"hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-Coder-GGUF") + "{q}/",
"file": "Qwen3.8-Flash-Next-GSQ-RCO-{q}-0000{i}-of-00002.gguf", "tag": "coder-",
"mmproj_hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-Coder-GGUF"),
"mmproj": "mmproj-Qwen3.8-Flash-Next-BF16.gguf", "name": "qwen3.8-flash-next-coder",
"profile": "expert-profile-coder.bin"},
# EXPERIMENTAL: Unsloth's UD-Q4_K_XL of the original model (docs/UNSLOTH_Q4.md): four shards, no images yet
"unsloth": {"title": "Qwen3.8-Flash-Next (Unsloth)", "by": "Unsloth's 4-bit quantization (EXPERIMENTAL)",
"about": "4-bit, 111 GB download, most experts read from the SSD: slow (7-8.5 tokens/s on a 64 GB PC)",
"hf": hf("unsloth/Qwen3.8-Flash-Next-GGUF") + "{q}/",
"file": "Qwen3.8-Flash-Next-{q}-0000{i}-of-00004.gguf", "shards": 4, "tag": "unsloth-",
"mmproj_hf": hf("ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF"),
"mmproj": "mmproj-Qwen3.8-Flash-Next-BF16.gguf", "name": "qwen3.8-flash-next-unsloth",
"experimental": True, "vision": False, "pack_args": ["--compat-bf16"], "sha256": UNSLOTH_SHARDS},
}
MMPROJ = "mmproj-Qwen3.8-Flash-Next-BF16.gguf"
# EXPERIMENTAL, off by default (setup asks): a control vector shipped with the repository, see its README
ESP_VECTOR = ROOT / "data" / "experimental-speed-projection" / "Qwen3.8-Flash-Next-experimental-speed-projection.gguf"
# the image encoder on the GPU (~1.2 GB at 1024 image tokens) warms up before the engine starts, so the engine
# sizes its expert slots around it and the default reserve (700 MiB) is enough; engines before 0.1.2 need more
VISION = {"gpu": {"max_tokens": 1024, "reserve_mib": 700},
"cpu": {"max_tokens": 300, "reserve_mib": 700}}
EXE = "strata.exe" if WIN else "strata"
VEXE = "strata-vision.exe" if WIN else "strata-vision"
# ------------------------------------------------------------------------------------------------ output
def say(msg=""):
print(msg, flush=True)
def step(n, title):
say()
say(f"=== Step {n}: {title} ===")
def ok(msg):
say(f" [ok] {msg}")
def warn(msg):
say(f" [!] {msg}")
def fail(msg, hint=None):
say(f"\n [X] {msg}")
if hint:
say(f" {hint}")
say("\nSetup stopped. Fix the item above and run it again - everything already done is kept and skipped.")
sys.exit(1)
def ask(question, choices, default, yes):
if yes:
return default
while True:
try:
a = input(f"{question} [{default}]: ").strip()
except EOFError:
fail("input ended before a setup answer was received",
"run setup in a terminal, or pass --yes to accept the recommended answers")
if not a:
return default
if a.lower() in [c.lower() for c in choices]:
return next(c for c in choices if c.lower() == a.lower())
say(f" please answer one of: {', '.join(choices)}")
def run(cmd, cwd=None, env=None, check=True, quiet=False):
say(" > " + " ".join(str(c) for c in cmd))
r = subprocess.run([str(c) for c in cmd], cwd=cwd, env=env,
stdout=subprocess.PIPE if quiet else None, stderr=subprocess.STDOUT if quiet else None,
text=True)
if check and r.returncode != 0:
if quiet and r.stdout:
say(r.stdout[-4000:])
fail(f"command failed (exit {r.returncode}): {Path(str(cmd[0])).name}")
return r
def out(cmd):
try:
return subprocess.run(cmd, capture_output=True, text=True, timeout=60).stdout
except (OSError, subprocess.TimeoutExpired):
return ""
def done(path: Path) -> bool:
"""A step's finish mark: <path>.done exists (written only after the step completed)."""
return path.with_name(path.name + ".done").exists()
def mark(path: Path, text=""):
path.with_name(path.name + ".done").write_text(text or time.strftime("%Y-%m-%d %H:%M"), encoding="utf-8")
# ------------------------------------------------------------------------------------------------ the PC
def _memory_status():
"""Windows' GlobalMemoryStatusEx: RAM, and the commit limit (ullTotalPageFile = RAM + page file)."""
class MS(ctypes.Structure):
_fields_ = [("dwLength", ctypes.c_ulong), ("dwMemoryLoad", ctypes.c_ulong),
("ullTotalPhys", ctypes.c_ulonglong), ("ullAvailPhys", ctypes.c_ulonglong),
("ullTotalPageFile", ctypes.c_ulonglong), ("ullAvailPageFile", ctypes.c_ulonglong),
("ullTotalVirtual", ctypes.c_ulonglong), ("ullAvailVirtual", ctypes.c_ulonglong),
("ullAvailExtendedVirtual", ctypes.c_ulonglong)]
m = MS()
m.dwLength = ctypes.sizeof(MS)
ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(m))
return m
def ram_gb():
if WIN:
return _memory_status().ullTotalPhys / 2**30
for line in open("/proc/meminfo"):
if line.startswith("MemTotal"):
return int(line.split()[1]) * 1024 / 2**30
return 0.0
def page_file_gb():
"""The page file's current size (GB) on Windows, None elsewhere. The graphics card's memory needs room there
too: under Windows' driver model every allocation on the card is also charged to the commit (RAM + page file),
so with the page file off or tiny the engine cannot use the free VRAM (issue #60)."""
if not WIN:
return None
m = _memory_status()
return max(0.0, (m.ullTotalPageFile - m.ullTotalPhys) / 2**30)
def cpu_info():
"""(name, avx2, avx512): avx512 means everything Strata's fast AVX-512 kernels use (F, BW, VL, VNNI, VBMI),
the same test the engine makes (cpu_avx512_ok), not just AVX-512F."""
name, avx2, avx512 = platform.processor() or "unknown CPU", False, False
if WIN:
pf = ctypes.windll.kernel32.IsProcessorFeaturePresent
avx2 = bool(pf(40)) or _cpuid_avx2() # PF_AVX2_INSTRUCTIONS_AVAILABLE, else the CPU itself (#159)
n = out(["powershell", "-NoProfile", "-Command", "(Get-CimInstance Win32_Processor).Name"]).strip()
name = n or name
avx512 = bool(pf(41)) and _cpuid_avx512_full()
else:
try:
txt = open("/proc/cpuinfo").read()
flags = set(re.search(r"^flags\s*:\s*(.*)$", txt, re.M).group(1).split())
avx2 = "avx2" in flags
avx512 = {"avx512f", "avx512bw", "avx512vl", "avx512_vnni", "avx512vbmi"} <= flags
m = re.search(r"^model name\s*:\s*(.*)$", txt, re.M)
name = m.group(1) if m else name
except OSError:
pass
return name, avx2, avx512
def _cpuid_avx512_full() -> bool:
"""Windows has no feature bit for VNNI / VBMI: ask the CPU (CPUID leaf 7) through a tiny machine-code stub."""
try:
code = bytes([0x53, 0x49, 0x89, 0xC8, 0xB8, 0x07, 0x00, 0x00, 0x00, 0x31, 0xC9, 0x0F, 0xA2, # push rbx; r8=rcx; cpuid(7,0)
0x41, 0x89, 0x18, 0x41, 0x89, 0x48, 0x04, 0x5B, 0xC3]) # [r8]=ebx,[r8+4]=ecx; pop rbx
k32 = ctypes.windll.kernel32
k32.VirtualAlloc.restype = ctypes.c_void_p
buf = k32.VirtualAlloc(None, len(code), 0x3000, 0x40)
if not buf:
return False
ctypes.memmove(buf, code, len(code))
regs = (ctypes.c_uint32 * 2)()
ctypes.CFUNCTYPE(None, ctypes.c_void_p)(buf)(ctypes.addressof(regs))
ebx, ecx = regs[0], regs[1]
need_ebx = (1 << 16) | (1 << 30) | (1 << 31) # F, BW, VL
need_ecx = (1 << 1) | (1 << 11) # VBMI, VNNI
return (ebx & need_ebx) == need_ebx and (ecx & need_ecx) == need_ecx
except Exception:
return False
def _run_stub(code: bytes, *args) -> None:
"""Runs a few bytes of x64 machine code (Windows calling convention: the arguments in rcx, rdx)."""
k32 = ctypes.windll.kernel32
k32.VirtualAlloc.restype = ctypes.c_void_p
k32.VirtualFree.argtypes = (ctypes.c_void_p, ctypes.c_size_t, ctypes.c_uint32)
buf = k32.VirtualAlloc(None, len(code), 0x3000, 0x40)
if not buf:
raise OSError("VirtualAlloc failed")
try:
ctypes.memmove(buf, code, len(code))
ctypes.CFUNCTYPE(None, *[ctypes.c_void_p] * len(args))(buf)(*args)
finally:
k32.VirtualFree(buf, 0, 0x8000)
def _cpuid_avx2() -> bool:
"""AVX2 asked from the CPU (CPUID leaf 7 EBX bit 5), with the OS saving the YMM registers (OSXSAVE + XCR0):
Windows' IsProcessorFeaturePresent(PF_AVX2) says no on some PCs whose CPU has it (a Ryzen 9 3950X, #159)."""
try:
def cpuid(leaf):
regs = (ctypes.c_uint32 * 4)()
_run_stub(bytes([0x53, 0x49, 0x89, 0xC8, 0x89, 0xD0, 0x31, 0xC9, 0x0F, 0xA2, # push rbx; r8=rcx; eax=edx; ecx=0; cpuid
0x41, 0x89, 0x00, 0x41, 0x89, 0x58, 0x04, 0x41, 0x89, 0x48, 0x08, # [r8]=eax, [r8+4]=ebx, [r8+8]=ecx
0x41, 0x89, 0x50, 0x0C, 0x5B, 0xC3]), # [r8+12]=edx; pop rbx
ctypes.addressof(regs), leaf)
return list(regs)
if cpuid(0)[0] < 7:
return False
ecx1 = cpuid(1)[2]
if not (ecx1 >> 27) & 1 or not (ecx1 >> 28) & 1: # OSXSAVE, AVX
return False
xcr0 = (ctypes.c_uint32 * 2)()
_run_stub(bytes([0x49, 0x89, 0xC8, 0x31, 0xC9, 0x0F, 0x01, 0xD0, # r8=rcx; ecx=0; xgetbv
0x41, 0x89, 0x00, 0x41, 0x89, 0x50, 0x04, 0xC3]), ctypes.addressof(xcr0))
if xcr0[0] & 6 != 6: # the OS saves XMM and YMM
return False
return bool((cpuid(7)[1] >> 5) & 1)
except Exception:
return False
def gpus():
"""Every NVIDIA GPU, numbered as nvidia-smi numbers them (by PCI bus, the order the engine is told to use)."""
s = out(["nvidia-smi", "--query-gpu=index,name,memory.total,compute_cap,driver_version",
"--format=csv,noheader,nounits"])
found = []
for line in s.strip().splitlines():
try:
idx, name, mem, cc, drv = [x.strip() for x in line.split(",")]
found.append({"index": int(idx), "name": name, "vram_gb": float(mem) / 1024.0, "arch": cc.replace(".", ""),
"driver": drv})
except ValueError:
continue
return found
GPU_PICK = None # --gpu N (issue #51); None: the card with the most VRAM
SPLIT_MIN_VRAM_GB = 8 # a card sharing a model holds the dense weights and its own
# prompt buffers too (docs/MULTI_GPU.md)
SPLIT_PROMPT_VRAM_GB = 12 # #448: a split stage lends a prompt chunk's buffers from its
# own cache; below this one cannot fund a 4096-token chunk
# (a 10 GB RTX 3080 beside a 32 GB card: 512 tokens, prompts
# 6.2x slower), while the big card alone could
def cc(g) -> str:
return f"{g['arch'][:-1]}.{g['arch'][-1]}"
def experimental_sm60() -> bool:
"""#295: STRATA_EXPERIMENTAL_SM60=1 admits Pascal (6.x) and Volta (7.0) cards: the community build
(-DSTRATA_EXPERIMENTAL_SM60=ON, compiled here with a CUDA 12.x toolkit), not the ready-made engine."""
return os.environ.get("STRATA_EXPERIMENTAL_SM60", "").strip() == "1"
def sm60_card(arch) -> bool:
return 60 <= int(arch) <= 70
def gpu_problem(g, together=False):
"""Why Strata cannot use this card, in plain words (None: it can)."""
if int(g["arch"]) < 75 and not (sm60_card(g["arch"]) and experimental_sm60()):
return (f"not supported - older than the RTX 20 series (compute capability {cc(g)}; Strata needs 7.5 or "
"newer" + ("; STRATA_EXPERIMENTAL_SM60=1 tries the community build for it" if sm60_card(g["arch"])
else "") + ")")
if together and g["vram_gb"] < SPLIT_MIN_VRAM_GB - 0.5:
return (f"not supported together with other GPUs - {g['vram_gb']:.0f} GB of VRAM (a card sharing the model "
f"needs {SPLIT_MIN_VRAM_GB} GB or more)")
return None
def gpu_rank(g):
"""The order cards share a model in: the newest generation first (it gets the first layers and most of the
work), then the most VRAM."""
return (-int(g["arch"]), -round(g["vram_gb"]), g["index"])
def gpu_name(g) -> str:
return f"GPU {g['index']} ({g['name']}, {g['vram_gb']:.0f} GB)"
def gpu_table(found) -> None:
say(" Your NVIDIA GPUs:")
for g in found:
p = gpu_problem(g)
say(f" GPU {g['index']}: {g['name']}, {g['vram_gb']:.0f} GB VRAM - " + ("can be used" if p is None else p))
def together_ok(found) -> list:
"""The cards that can share one model, in the order they would (empty if fewer than two)."""
ok_ = sorted([g for g in found if gpu_problem(g, together=True) is None], key=gpu_rank)
return ok_ if len(ok_) >= 2 else []
def split_short(cards) -> list:
"""#448: the later cards of a split that would cap its prompt chunk below what the first card alone reads (a card
under SPLIT_PROMPT_VRAM_GB beside one that has it). Empty: the split is recommended as before."""
if len(cards) < 2 or cards[0]["vram_gb"] < SPLIT_PROMPT_VRAM_GB - 0.5:
return []
return [g for g in cards[1:] if g["vram_gb"] < SPLIT_PROMPT_VRAM_GB - 0.5]
def split_short_note(g) -> str:
return (f"GPU {g['index']} ({g['name']}, {g['vram_gb']:.0f} GB) is too small to lend a split its prompt buffers: "
"it would cap prompt reading at 512-2048-token chunks, several times slower than the first card alone "
"(#448). It can serve as a helper expert cache instead (docs/SECOND_GPU.md)")
def parse_gpus(text, found) -> list:
"""--gpus / --gpu with several: "0,2" or "all" (every card that can share the model)."""
if str(text).strip().lower() == "all":
sel = [g["index"] for g in together_ok(found)]
if not sel:
gpu_table(found)
fail("--gpus all: this PC does not have two GPUs Strata can use together")
return sel
try:
sel = [int(x) for x in str(text).split(",") if x.strip()]
except ValueError:
fail(f"--gpus takes GPU numbers as nvidia-smi numbers them, e.g. --gpus 0,2 (or --gpus all), not {text!r}")
if len(sel) < 2 or len(set(sel)) != len(sel):
fail("--gpus takes two or more different GPUs, e.g. --gpus 0,2 (one GPU: --gpu 0)")
return sel
def check_gpus(sel, found, what="", yes=False, named=False) -> None:
"""Stops with a plain message when a chosen card is missing or cannot be used, and says what can. named: the user
named these cards (--gpus 0,1, or a config that has them): a card that is only short of VRAM for sharing the model
is then a risk to confirm, not a stop (the owner's rule; --yes with the named cards is the consent)."""
together = len(sel) > 1
for i in sel:
g = next((x for x in found if x["index"] == i), None)
p = "not found on this PC" if g is None else gpu_problem(g, together)
if p is None:
continue
if named and g is not None and gpu_problem(g) is None: # it runs Strata; only its VRAM is small
confirm_risk(f"GPU {i} ({g['name']}) has {g['vram_gb']:.0f} GB of VRAM: a card sharing the model needs "
f"{SPLIT_MIN_VRAM_GB} GB or more (it holds the dense weights of its layers and its own prompt "
"buffers), so the model may not start, or run slower than without it", True, yes,
f"GPU {i} ({g['name']}) {what}is not used together with other GPUs: {p}",
"leave it out of --gpus, or answer y to use it anyway", " Use it anyway?")
warn(f"GPU {i} ({g['name']}) is used together with the others, as you chose")
continue
say()
gpu_table(found)
can = together_ok(found)
single = [x for x in found if gpu_problem(x) is None]
ones = " or ".join(f"--gpu {x['index']}" for x in single)
both = "--gpus " + ",".join(str(x["index"]) for x in can) if can else ""
hint = ((f"use these together: {both}" + (f" (or one card: {ones})" if not together else "")) if can else
f"use one card: {ones}" if single else "Strata needs an NVIDIA RTX 20 series or newer card")
fail(f"GPU {i}{'' if g is None else ' (' + g['name'] + ')'} {what}cannot be used: {p}", hint)
def engine_archs():
"""The GPU generations the installed engine has code for: (archs, ptx), or None when there is none."""
info = ROOT / "engine" / "BUILD.json"
try:
meta = json.loads(info.read_text())
except (OSError, ValueError):
return None
return [int(x) for x in meta.get("archs", [])], bool(meta.get("ptx"))
def engine_archs_hip():
"""The AMD architectures the installed HIP engine was compiled for ("gfx1201", ...), or None."""
try:
meta = json.loads((ROOT / "engine" / "BUILD.json").read_text())
except (OSError, ValueError):
return None
return [str(x) for x in meta.get("archs", [])] if meta.get("backend") == "hip" else None
def engine_runs_on(g) -> bool:
ea = engine_archs()
if ea is None or not ea[0]:
return True
archs, ptx = ea
return int(g["arch"]) in archs or (ptx and int(g["arch"]) > max(archs))
def start_gpus(text):
"""--gpus when starting an installed model: NVIDIA cards as nvidia-smi numbers them, or on a PC whose AMD cards
are the ones Strata can use, AMD cards as setup lists them ("all": every supported AMD card)."""
if not text:
return None
if str(text).strip().lower() == "all" and not WIN and not together_ok(gpus()):
amd = amd_gpus()
if len([g for g in amd if amd_problem(g) is None]) >= 2:
return [g["index"] for g in amd_parse_gpus("all", amd)]
return parse_gpus(text, gpus())
def choose_gpus(a, found) -> list:
"""Which cards this install uses: --gpus / --gpu, or asked when two or more can share the model (the two best
together recommended), else the supported card with the most VRAM. Returns their numbers, the main one first."""
if a.gpus:
sel = parse_gpus(a.gpus, found)
check_gpus(sel, found, yes=a.yes, named=str(a.gpus).strip().lower() != "all")
return sel
if a.gpu is not None:
check_gpus([a.gpu], found)
return [a.gpu]
single = sorted([g for g in found if gpu_problem(g) is None], key=lambda x: (-round(x["vram_gb"]), x["index"]))
if not single:
gpu_table(found)
fail("none of your GPUs can run Strata", "it needs an NVIDIA RTX 20 series or newer (compute capability 7.5+)")
can = together_ok(found)
if not can:
return [single[0]["index"]]
say()
say(f" Strata can run the model on one GPU, or share it across {'these' if len(can) > 2 else 'both'}: then each"
" card holds the")
say(" experts of its own layers, so together they hold about twice as many, and prompts are read about 20%")
say(" faster (details: docs/MULTI_GPU.md). A much slower extra card can also make it slower.")
opts = [can[:2]] + ([can] if len(can) > 2 else []) + [[g] for g in single]
# #448: a pair whose second card cannot lend a 4096-token chunk recommends the first card alone (still offered)
short = split_short(can[:2])
rec = next(i for i, o in enumerate(opts, 1) if o == [can[0]]) if short else 1
for i, o in enumerate(opts, 1):
label = (" + ".join(gpu_name(g) for g in o) + " together") if len(o) > 1 else gpu_name(o[0]) + " only"
say(f" {i}) {label}" + (" (recommended)" if i == rec else ""))
for g in found:
if gpu_problem(g, together=True) is not None:
say(f" (GPU {g['index']}, {g['name']}: {gpu_problem(g, together=True)})")
for g in short:
say(f" ({split_short_note(g)})")
pick = opts[int(ask("Which GPUs?", [str(i) for i in range(1, len(opts) + 1)], str(rec), a.yes or a.check)) - 1]
return [g["index"] for g in pick]
def split_mmap(cfg: dict) -> bool:
"""#364 #384: the low-RAM mode's resident variant (--resident-experts) has no layer split yet. A config with it
that runs on several GPUs reads the experts the GPUs do not hold through the OS file cache instead
(--mmap-experts: the placement those reports measured 1.3-1.6x faster than one GPU), said plainly - the engine
used to refuse the pair. True when the config changed."""
a = cfg.get("args", [])
if "--resident-experts" not in a:
return False
a[a.index("--resident-experts")] = "--mmap-experts"
warn("the low-RAM mode's resident variant (--resident-experts) has no layer split yet: on several GPUs the experts "
"the GPUs do not hold are read through the OS file cache (--mmap-experts) instead, and RAM can fill up to 0 "
"free during long prompts. One GPU keeps them in RAM (steady RAM use): START-HERE --setup, or --gpu N for a "
"start")
return True
def unsloth_split_need_gb(model="UD-Q4_K_XL") -> float:
"""#498: the RAM UD-Q4_K_XL needs on several GPUs, where it has no RAM budget (the engine refuses
--resident-budget-gib with a layer split): its GGUF files and UNSLOTH_RAM_LEFT_GB more (~135 GB). Measured safe
at 165 GiB (2x RTX 3090: MemAvailable never under 68 GiB); the 0-free case of #384 was 47 GB with a 70 GB model."""
return MODELS[model]["download_gb"] + UNSLOTH_RAM_LEFT_GB
def split_budget(cfg: dict) -> bool:
"""#498: a UD-Q4_K_XL config (its RAM budget, --resident-budget-gib) started on several GPUs. The engine refuses
the budget with a layer split (it exited with code 2), so the split runs without it - all the experts loaded into
RAM at start - where the RAM holds the GGUFs and 24 GB more; else setup stops, before the config is saved. True
when the config changed."""
a = cfg.get("args", [])
if "--resident-budget-gib" not in a:
return False
need, ram = unsloth_split_need_gb(), ram_gb()
if ram < need:
fail(f"UD-Q4_K_XL cannot share its RAM budget across GPUs (the engine has no layer split with it), and without "
f"the budget it needs ~{need:.0f} GB of RAM (its GGUF files and {UNSLOTH_RAM_LEFT_GB} GB more); this PC "
f"has {ram:.0f} GB", "start it on one GPU: START-HERE.bat --gpu N (Linux: ./setup.sh --gpu N)")
i = a.index("--resident-budget-gib")
del a[i:i + 2]
ok("UD-Q4_K_XL on several GPUs: no RAM budget (the engine has none with a layer split) - all its experts are "
"loaded into RAM from the model files at start, and the files pass through the OS file cache (#498)")
return True
def offer_together(cfg_path: Path, cfg: dict, yes: bool) -> dict:
"""Starting a model set up for one card on a PC with two or more that can share it: asked once (the answer is
saved in its config)."""
if isinstance(cfg.get("gpu"), list) or cfg.get("gpus_asked"):
return cfg
found = gpus()
can = together_ok(found)
if not can:
return cfg
# #498: UD-Q4_K_XL's RAM budget has no layer split; without it the RAM must hold the GGUFs and 24 GB more
budget = "--resident-budget-gib" in cfg.get("args", [])
if budget and ram_gb() < unsloth_split_need_gb():
return cfg
pair = can[:2]
cfg["gpus_asked"] = True
# #364 #384: the resident low-RAM variant stays on one card unless the user says otherwise (its RAM use is steady)
resident = "--resident-experts" in cfg.get("args", [])
say()
say(" This PC has " + " and ".join(gpu_name(g) for g in pair) + ": Strata can share the model across both.")
say(" Together they hold about twice the model's experts and read prompts about 20% faster (docs/MULTI_GPU.md).")
if resident:
say(" This model runs in the low-RAM mode with its experts kept in RAM, on one GPU (recommended: steady RAM")
say(" use). On both, the experts the GPUs do not hold are read through the OS file cache instead: faster in")
say(" two reports (#364, #384), but RAM can fill up to 0 free during long prompts.")
if budget:
say(" This model (UD-Q4_K_XL) runs on one GPU with a RAM budget of its experts (recommended: the tested")
say(" setup). On both it has no budget: all its experts are loaded into RAM at start, which this PC's RAM")
say(" holds - about twice as fast in #498 (2x RTX 3090: 31 -> 64-78 tokens/s).")
short = split_short(pair) # #448: one card recommended (asked "n" by default), as for --resident
for g in short:
say(f" {split_short_note(g)}.")
missing = [g for g in pair if not engine_runs_on(g)]
if missing:
say(" The installed engine has no code for " + ", ".join(g["name"] for g in missing) + ": to use them "
"together, run START-HERE.bat --setup --gpus " + ",".join(str(g["index"]) for g in pair))
elif ask(" Use both from now on? (you can change it later: START-HERE.bat --gpu N for one card)",
["y", "n"], "n" if resident or short or budget else "y", yes) == "y":
cfg["gpu"] = [g["index"] for g in pair]
cfg["layer_split"] = cfg.get("layer_split") or "auto"
split_mmap(cfg)
split_budget(cfg)
ok("from now on this model runs on " + " + ".join(gpu_name(g) for g in pair))
else:
ok("staying on one GPU (START-HERE.bat --gpus " + ",".join(str(g["index"]) for g in pair) + " switches)")
write_config(cfg_path, cfg)
return cfg
def gpu_info(pick=None):
"""The GPU Strata runs on: `pick` (nvidia-smi's number) if given, else the one with the most VRAM (ties: the
lower number). None when there is no NVIDIA GPU. The dict also says how many there are ("count")."""
found = gpus()
if not found:
return None
pick = GPU_PICK if pick is None else pick
if pick is not None:
g = next((x for x in found if x["index"] == pick), None)
if g is None:
fail(f"there is no GPU {pick}: " + ", ".join(f"{x['index']} = {x['name']}" for x in found))
else:
g = max(found, key=lambda x: (round(x["vram_gb"]), -x["index"]))
return {**g, "count": len(found)}
def find_nvcc(below=None):
"""The newest CUDA toolkit's nvcc and its (major, minor); with `below`, the newest older than that version."""
cands = [shutil.which("nvcc")]
if os.environ.get("CUDA_PATH"):
cands.append(str(Path(os.environ["CUDA_PATH"]) / "bin" / ("nvcc.exe" if WIN else "nvcc")))
if WIN:
base = Path(r"C:\Program Files\NVIDIA GPU Computing Toolkit\CUDA")
if base.exists():
cands += [str(p / "bin" / "nvcc.exe") for p in sorted(base.iterdir(), reverse=True)]
else:
cands += [str(p / "bin" / "nvcc") for p in sorted(Path("/usr/local").glob("cuda*"), reverse=True)]
cands += [str(p / "bin" / "nvcc") for p in sorted(Path("/opt").glob("cuda*"), reverse=True)] # Arch (#46)
best = (None, None)
for c in dict.fromkeys(cands): # every toolkit found; the newest wins
if c and Path(c).exists():
v = re.search(r"release (\d+)\.(\d+)", out([c, "--version"]))
ver = (int(v.group(1)), int(v.group(2))) if v else None
if ver and (below is None or ver < below) and (best[1] is None or ver > best[1]):
best = (c, ver)
return best
def find_vcvars():
vswhere = Path(os.environ.get("ProgramFiles(x86)", r"C:\Program Files (x86)")) / "Microsoft Visual Studio/Installer/vswhere.exe"
if not vswhere.exists():
return None
# CUDA 13 accepts Visual Studio 2019 and 2022 only: a newer one (2026 = version 18) installed next to them
# must not be picked ("unsupported Microsoft Visual Studio version"); with only a newer one there is none
p = out([str(vswhere), "-latest", "-products", "*", "-version", "[16.0,18.0)", "-requires",
"Microsoft.VisualStudio.Component.VC.Tools.x86.x64", "-property", "installationPath"]).strip()
v = Path(p) / "VC/Auxiliary/Build/vcvars64.bat" if p else None
return v if v and v.exists() else None
def find_tool(name):
"""A tool on PATH, or the one pip installed next to this Python (cmake, ninja)."""
p = shutil.which(name)
if p:
return p
for d in (Path(sys.executable).parent / "Scripts", Path(sys.executable).parent,
Path.home() / ".local" / "bin"):
c = d / (name + (".exe" if WIN else ""))
if c.exists():
return str(c)
return None
def free_gb(path):
path.mkdir(parents=True, exist_ok=True)
return shutil.disk_usage(path).free / 1e9
# ------------------------------------------------------------------------------------------------ downloads
def drop_archive(z: Path) -> None:
"""An unpacked or refused engine archive and its .done mark go: a refused one kept them, and every later run
reused it ("already downloaded") instead of the published one (PR #324)."""
z.unlink(missing_ok=True)
z.with_name(z.name + ".done").unlink(missing_ok=True)
def download(url, dst: Path, what=None):
"""Resumable HTTP(S) download with a progress line; `file://` and plain paths are copied (tests, mirrors).
A finished file gets a <name>.done mark, so a later run skips it without asking the server."""
dst.parent.mkdir(parents=True, exist_ok=True)
if dst.exists() and done(dst):
ok(f"{what or dst.name} already downloaded")
return
if not url.startswith(("http://", "https://")):
src = Path(url[7:] if url.startswith("file://") else url)
if not src.exists():
fail(f"not found: {src}")
shutil.copyfile(src, dst)
mark(dst)
ok(f"{what or dst.name} copied")
return
part = dst.with_name(dst.name + ".part")
total = 0
for attempt in range(5):
try:
req = urllib.request.Request(url, method="HEAD", headers={"User-Agent": "strata-setup"})
total = int(urllib.request.urlopen(req, timeout=60).headers.get("Content-Length", 0))
break
except urllib.error.HTTPError as e:
if e.code == 404 and hf_unpinned(url) != url: # #214: the pinned revision is gone from the repository
warn(f"{what or dst.name}: not at the pinned revision any more; downloading the repository's "
"current file")
url = hf_unpinned(url)
continue
if attempt == 4:
fail(f"cannot reach {url.split('/')[2]} ({e})", "check your internet connection and run it again")
time.sleep(5)
except OSError as e:
if attempt == 4:
fail(f"cannot reach {url.split('/')[2]} ({e})", "check your internet connection and run it again")
time.sleep(5)
if dst.exists() and total and dst.stat().st_size == total: # finished by an older setup (no mark yet)
mark(dst)
ok(f"{what or dst.name} already downloaded")
return
have = part.stat().st_size if part.exists() else 0
for attempt in range(30):
try:
req = urllib.request.Request(url, headers={"User-Agent": "strata-setup", "Range": f"bytes={have}-"})
with urllib.request.urlopen(req, timeout=60) as r, open(part, "ab" if have else "wb") as f:
if have and r.status != 206: # the server ignored the range: start over
f.seek(0)
f.truncate()
have = 0
last = 0.0
while True:
b = r.read(8 << 20)
if not b:
break
f.write(b)
have += len(b)
if time.time() - last > 2:
last = time.time()
size = f"{have / 1e9:6.2f} / {total / 1e9:.2f} GB ({100 * have / total:.0f}%)" if total \
else f"{have / 1e6:7.1f} MB"
print(f"\r {what or dst.name}: {size} ", end="", flush=True)
print()
if not total or have >= total:
break
except OSError as e:
print()
warn(f"download interrupted ({e}); retrying in 10 s ...")
time.sleep(10)
if total and part.stat().st_size != total:
fail(f"could not finish downloading {dst.name}: {part.stat().st_size:,} bytes on disk, the server says {total:,}",
"check your internet connection and run it again (the download resumes where it stopped)")
part.replace(dst)
mark(dst)
ok(f"{what or dst.name} downloaded")
def whole_shard(s: Path) -> bool:
"""A shard as long as its own tensor directory says (check_shards' test, without stopping setup)."""
sys.path.insert(0, str(ROOT / "tools"))
from gguf_reader import GGUFFile
try:
g = GGUFFile(s)
return s.stat().st_size >= g.data_start + max((t.offset + (t.expected_bytes() or 0) for t in g.tensors),
default=0)
except (OSError, ValueError, struct.error):
return False
SHARD_NAME = re.compile(r"-(\d{5})-of-(\d{5})\.gguf$")
def n_shards(fam: dict, model: str) -> int:
"""🍷 the shard count: the size's own (the Winery tiers differ), else the family's."""
return int(MODELS.get(model, {}).get("shards", fam.get("shards", 2)))
def shard_file(fam: dict, model: str, i: int) -> str:
return fam["file"].format(q=model, i=i, n=n_shards(fam, model))
def gguf_dir_shards(folder: Path, fam: dict, model: str) -> list[Path]:
"""--gguf-dir's shards (#305): every -0000i-of-0000N file of the model, N read from the first shard's name as the
engine and tools/iq_pack.py do. The published name first; else the one first shard in the folder whose name has
the size in it (an upload split or named differently: -00001-of-00003, Unsloth's ...-00001-of-00004.gguf). A
missing shard is check_shards' error later, as before."""
first = folder / shard_file(fam, model, 1)
if not first.exists():
found = sorted(p for p in folder.glob("*-00001-of-*.gguf") if SHARD_NAME.search(p.name))
mine = [p for p in found if model.lower() in p.name.lower()]
pick = mine if mine else found
if len(pick) == 1:
first = pick[0]
m = SHARD_NAME.search(first.name)
total = int(m.group(2)) if m else 1
if not m or total < 1:
return [first]
stem = first.name[:m.start()]
return [first.with_name("%s-%05d-of-%05d.gguf" % (stem, i, total)) for i in range(1, total + 1)]
# #444: the quantization in a GGUF's name (Unsloth's UD-IQ3_XXS, a K-quant's Q2_K_XL, a GSQ-RCO IQ3_S, ...)
GGUF_QUANT = re.compile(r"(?<![A-Za-z0-9])((?:UD-)?(?:I?Q\d+(?:_[A-Za-z0-9]+)*|BF16|F16|F32))"
r"(?=-\d{5}-of-\d{5}\.gguf$|\.gguf$)", re.I)
SUPPORTED_GGUFS = ("Strata runs ISTA-DASLab's GSQ-RCO files (Qwen3.8-Flash-Next Q2_0, IQ2_XS, IQ3_XXS, IQ3_S; Swift "
"1.5's; the Coder's IQ1_M) and Unsloth's UD-Q4_K_XL only: other GGUFs (Unsloth's UD-IQ3_XXS or "
"UD-Q2_K_XL, K-quants) cannot be used")
def gguf_unsupported(name: str) -> str | None:
"""#444: the quantization a GGUF's name says, when it is one Strata cannot run (not a setup size); else None."""
m = GGUF_QUANT.search(name)
return m.group(1) if m and m.group(1).upper() not in MODELS and not name.lower().startswith("mmproj") else None
def gguf_choice(name: str) -> tuple | None:
"""#444: (--family, --model) whose published first shard this file is, or None (the Coder's IQ1_M is named like
the original's sizes: the size tells them apart)."""
for f, d in FAMILIES.items():
for m in MODELS:
if f in MODELS[m].get("families", ("qwen", "swift")) and name == shard_file(d, m, 1):
return f, m
return None
def gguf_dir_problem(folder: Path, first: Path, fam: dict, model: str) -> tuple | None:
"""#444: (message, hint) when --gguf-dir has no file setup can use for this choice: the chosen shard is a GGUF
Strata cannot run (Unsloth's UD-IQ3_XXS taken for IQ3_XXS by its name), or it is missing and the folder holds
other GGUFs - then the hint names the --family/--model of the usable ones. None otherwise (a missing shard in a
folder without GGUFs stays check_shards' "missing")."""
bad = gguf_unsupported(first.name) if first.exists() else None
if bad:
return f"{first.name} is {bad}, a GGUF Strata cannot run", SUPPORTED_GGUFS
if first.exists():
return None
firsts = sorted(p.name for p in folder.glob("*.gguf")
if not p.name.lower().startswith("mmproj") and (not SHARD_NAME.search(p.name)
or SHARD_NAME.search(p.name).group(1) == "00001"))
usable = list(dict.fromkeys(c for c in map(gguf_choice, firsts) if c))
unusable = [n for n in firsts if gguf_unsupported(n)]
if not usable and not unusable:
return None
hint = SUPPORTED_GGUFS
if unusable:
hint += ".\n Not usable here: " + ", ".join(unusable)
if usable:
hint += ".\n Usable here: " + ", ".join(f"--family {f} --model {m}" for f, m in usable)
return f"{folder} has no {fam['title']} {model} file", hint
def verify_sha256(s: Path, size: int, sha: str) -> None:
"""A shard's size and SHA-256 against the pinned values (the Unsloth file); the result is kept in its finish mark,
so the ~5 minutes of hashing 111 GB happen once. A wrong file is deleted, so the next run downloads it again."""
m = s.with_name(s.name + ".done")
if m.exists() and f"sha256 {sha}" in m.read_text(encoding="utf-8", errors="replace"):
return
have = s.stat().st_size if s.exists() else -1
if have != size:
fail(f"{s.name} is {have:,} bytes, not {size:,}", "delete it and run setup again (the download restarts)")
say(f" checking {s.name} (SHA-256, {size / 1e9:.1f} GB) ...")
h = hashlib.sha256()
with open(s, "rb") as f:
while True:
b = f.read(16 << 20)
if not b:
break
h.update(b)
if h.hexdigest() != sha:
s.unlink(missing_ok=True)
m.unlink(missing_ok=True)
fail(f"{s.name} has the wrong SHA-256 ({h.hexdigest()}, expected {sha}): deleted",
"run setup again to download it again")
mark(s, f"sha256 {sha}")
def resident_budget_gib(model, ram, kv_ram_gb=0.0) -> int:
"""UD-Q4_K_XL: the GiB of experts the engine keeps in RAM (--resident-budget-gib): the RAM (GiB, ram_gb()) less
24 for the OS, the engine and the file cache the other experts are read through, less a KV cache streamed to
RAM; at most all of them, at least 8. 64 GB: 40, the measured setting (docs/UNSLOTH_Q4.md)."""
gib = round(ram) - UNSLOTH_RAM_LEFT_GB - math.ceil(kv_ram_gb)
return max(8, min(gib, int(MODELS[model]["arena_gb"] / 1.073741824)))
def budget_choice(model, ram, asked) -> float:
"""S4: UD-Q4_K_XL's RAM budget: --resident-budget-gib N as given, else the recommendation (resident_budget_gib).
More than the recommendation is kept, with what it risks (the owner's rule: setup recommends, it never forces)."""
rec = resident_budget_gib(model, ram)
if asked is None:
return rec
if asked > rec:
warn(f"a {asked:g} GiB RAM budget is more than setup recommends for this PC ({rec} GiB: the RAM less "
f"{UNSLOTH_RAM_LEFT_GB} GB for the OS, the engine and the file cache that reads the other experts). Kept "
"as you chose: the engine clamps it to the RAM it finds free at start (less 4 GB), and the file cache "
"gets less room - it may be slower, or run the PC out of RAM under load")
return int(asked) if asked == int(asked) else asked
def check_shards(shards):
"""Every shard present and whole, or setup stops naming the file and the numbers. Whole means as long as
its own tensor directory says (the header is read, the data is not): a truncated copy (--gguf-dir, a .part
renamed by hand, a download finished by an older setup) otherwise passes as a model file and the engine
fails much later, at the first tensor that runs past the end."""
sys.path.insert(0, str(ROOT / "tools"))
from gguf_reader import GGUFFile
for s in shards:
if not s.exists():
fail(f"missing {s}")
try:
g = GGUFFile(s)
except (ValueError, struct.error) as e:
fail(f"{s.name} is not a whole GGUF shard ({e})", "delete it and run setup again")
need = g.data_start + max((t.offset + (t.expected_bytes() or 0) for t in g.tensors), default=0)
have = s.stat().st_size
if have < need:
fail(f"{s.name} is short: {have:,} of {need:,} bytes ({need - have:,} missing)",
"delete it and run setup again (or copy the whole file into --gguf-dir)")
def get_llama_cpp():
"""llama.cpp at the pinned commit (ggml for the build, gguf-py for the tools, mtmd for images), as a zip: no git."""
llama = ROOT / "third_party" / "llama.cpp"
if (llama / "ggml" / "CMakeLists.txt").exists() and (llama / "gguf-py").is_dir():
return llama
z = ROOT / "third_party" / f"llama.cpp-{LLAMA_CPP_COMMIT[:7]}.zip"
download(LLAMA_CPP_ZIP, z, "llama.cpp source")
tmp = ROOT / "third_party" / "_unpack"
shutil.rmtree(tmp, ignore_errors=True)
with zipfile.ZipFile(z) as f:
# llama.cpp's own web UI (tools/ui) is not used, and its deep paths passed Windows' 260-character limit in a
# folder like Downloads\Strata-main\Strata-main (#206)
f.extractall(tmp, [m for m in f.namelist() if "/tools/ui/" not in m])
top = next(tmp.iterdir())
shutil.rmtree(llama, ignore_errors=True)
# PR #63: on Windows a rename can fail with PermissionError while an antivirus scanner still holds a file of the
# fresh unpack; shutil.move falls back to copy-and-delete, and a few retries let the scanner finish. The target
# is `llama` itself - moving into its parent would keep the zip's `llama.cpp-<sha>` folder name.
for attempt in range(5):
try:
shutil.move(str(top), str(llama))
break
except PermissionError:
if attempt == 4:
raise
shutil.rmtree(llama, ignore_errors=True) # a partial copy from the failed attempt
time.sleep(2)
shutil.rmtree(tmp, ignore_errors=True)
z.unlink(missing_ok=True)
z.with_name(z.name + ".done").unlink(missing_ok=True)
return llama
def req_name(line: str) -> str:
"""The distribution name of a requirement line ("numpy==2.5.3; python_version >= '3.12'" -> "numpy")."""
return re.split(r"[\s<>=!~;\[]", line.strip(), maxsplit=1)[0].lower().replace("_", "-")
def requirement_lines(path: Path | None = None) -> list[str]:
"""requirements.txt's requirements, comments and blank lines left out."""
lines = []
for raw in (path or REQUIREMENTS).read_text(encoding="utf-8").splitlines():
line = raw.split("#", 1)[0].strip()
if line:
lines.append(line)
return lines
def _installed(name: str) -> bool:
try:
import importlib.metadata as md
md.distribution(name)
return True
except Exception:
return False
def pip_install(packages, what):
"""pip install into .venv, skipped when the same list was installed before. An install from before the pinned
requirements (#214) recorded bare names: those packages are kept as they are (nothing is reinstalled), and the
pinned dependencies it already has count as installed."""
stamp = Path(sys.prefix) / ".strata-pip.json"
have = json.loads(stamp.read_text()) if stamp.exists() else []
bare = {p.lower() for p in have if req_name(p) == p.lower()}
need = [p for p in packages if p not in have and req_name(p) not in bare
and not (bare and "==" in p and _installed(req_name(p)))]
if not need:
ok(f"{what} already installed")
return
say(f" Installing {what} ...")
pip = [sys.executable, "-m", "pip", "install", "--quiet", "--disable-pip-version-check", "--prefer-binary"]
if run([*pip, *need], check=False).returncode != 0:
# no ready-made wheel of a pinned version for this Python (pip tried to compile, e.g. Pillow): take the
# newest release that has one instead of failing the whole setup
say(" A pinned package has no ready-made wheel for this Python; retrying with ready-made wheels only ...")
loose = sorted({req_name(p) for p in need})
run([*pip, "--only-binary", ":all:", *loose])
stamp.write_text(json.dumps(sorted(set(have) | set(need)), indent=0))
ok(f"{what} installed")
def cuda_lib_dirs():
"""Where pip put NVIDIA's CUDA libraries (nvidia/cu13/bin/x86_64 on Windows, nvidia/cu13/lib on Linux)."""
pattern = "cublas64_13.dll" if WIN else "libcublas.so.13*"
dirs = []
for sp in {Path(p) for p in sys.path if p.endswith("site-packages")}:
for hit in (sp / "nvidia").rglob(pattern) if (sp / "nvidia").is_dir() else []:
if hit.parent not in dirs:
dirs.append(hit.parent)
return [str(d) for d in dirs]
# ------------------------------------------------------------------------------------------------ AMD
# The RX 7900 XT / XTX (gfx1100) and the RX 9070 series / Radeon AI PRO R9700 (gfx1201) on Linux, through the HIP
# backend (docs/AMD_HIP.md); the RX 7800 XT / 7700 XT (gfx1101, #254) and the RX 9060 XT (gfx1200, #256) were run by
# their owners; the RX 6800 / 6900 series (gfx1030, #311) runs but is unvalidated. There is no ready-made AMD engine: ROCm comes from AMD's TheRock Python wheels into .venv (no sudo;
# a system ROCm 7 in /opt/rocm is used when it has hipcc and hipBLAS) and the engine is compiled here for the cards.
# No images yet.
ROCM_INDEXES = {"gfx1100": "https://rocm.nightlies.amd.com/v2/gfx110X-dgpu/", # TheRock's wheels per GPU family
"gfx1101": "https://rocm.nightlies.amd.com/v2/gfx110X-dgpu/",
"gfx1200": "https://rocm.nightlies.amd.com/v2/gfx120X-all/",
"gfx1201": "https://rocm.nightlies.amd.com/v2/gfx120X-all/",
"gfx1030": "https://rocm.nightlies.amd.com/v2/gfx103X-all/"}
ROCM_VERSION = os.environ.get("STRATA_ROCM_VERSION", "7.10.0a20251120") # what Strata's HIP build was tested with
ROCM_SYSTEM_MIN = (7, 0) # an older system ROCm is passed over for the wheels (gfx1201 needs ROCm 6.4 or newer)
AMD_ARCHS = ("gfx1100", "gfx1101", "gfx1200", "gfx1201", "gfx1030")
AMD_NAMES = {"gfx1100": "AMD Radeon RX 7900 series (gfx1100)", # when sysfs has no product name
"gfx1101": "AMD Radeon RX 7800 XT / 7700 XT (gfx1101)",
"gfx1200": "AMD Radeon RX 9060 series (gfx1200)",
"gfx1201": "AMD Radeon RX 9070 series / AI PRO R9700 (gfx1201)",
"gfx1030": "AMD Radeon RX 6800 / 6900 series (gfx1030)"}
AMD_CARDS = ("the RX 7900 XT / XTX (gfx1100), RX 7800 XT / 7700 XT (gfx1101), RX 9060 XT (gfx1200) and "
"RX 9070 / 9070 XT / Radeon AI PRO R9700 (gfx1201), and the RX 6800 / 6900 series (gfx1030, unvalidated)")
def rocm_index(arch):
return os.environ.get("STRATA_ROCM_INDEX") or ROCM_INDEXES[arch]
def amd_gpus(sysfs="/sys"):
"""AMD GPUs from the kernel's KFD topology (the amdgpu driver; no ROCm needed), numbered as HIP numbers them:
the GPU nodes in order, the CPU nodes skipped. Integrated GPUs are listed too (not supported).
sysfs: the tree to read (tools/test_setup_amd.py passes a mocked one). Windows: amd_gpus_win."""
if WIN:
return amd_gpus_win()
base = Path(sysfs) / "class/kfd/kfd/topology/nodes"
found = []
if not base.is_dir():
return found
for node in sorted((p for p in base.iterdir() if p.name.isdigit()), key=lambda p: int(p.name)):
try:
props = {}
for line in (node / "properties").read_text().splitlines():
k, _, v = line.partition(" ")
props[k] = v.strip()
ver = int(props.get("gfx_target_version") or 0)
if ver == 0 or int(props.get("simd_count") or 0) == 0:
continue
except (OSError, ValueError):
continue
arch = f"gfx{ver // 10000}{(ver // 100) % 100:x}{ver % 100:x}"
dev = Path(sysfs) / f"class/drm/renderD{props.get('drm_render_minor', '')}/device"
try:
vram = int((dev / "mem_info_vram_total").read_text()) / 2 ** 30
except (OSError, ValueError):
vram = 0.0
try:
name = (dev / "product_name").read_text().strip() or f"AMD Radeon ({arch})"
except OSError:
name = f"AMD Radeon ({arch})"
if name == f"AMD Radeon ({arch})" and arch in AMD_NAMES:
name = AMD_NAMES[arch]
found.append({"index": len(found), "name": name, "vram_gb": vram, "arch": arch, "driver": "amdgpu",
"vendor": "amd"})
return found
def amd_problem(g):
if g["arch"] not in AMD_ARCHS:
return f"not supported - Strata's AMD backend runs on {AMD_CARDS} only, this is {g['arch']}"
if g.get("cannot_run"): # Windows: the installed engine's own check (--list-devices)
return g["cannot_run"]
return None
def amd_gpus_win() -> list[dict]:
"""Windows: the AMD GPUs as the HIP runtime numbers them once the HIP engine is installed (hip_devices), else in
the display-adapter order (amd_gpus_windows)."""
return hip_devices() or amd_gpus_windows()
def amd_parse_gpus(text, amd) -> list:
"""--gpus with AMD cards, numbered as HIP numbers them (setup's list): "1,0", or "all" (every supported card, the
most VRAM first). Every chosen card must be one Strata supports (AMD_ARCHS; they may be of different
architectures: the engine is compiled for each). Returns the cards, the main one first."""
usable = [g for g in amd if amd_problem(g) is None]
if str(text).strip().lower() == "all":
sel = [g["index"] for g in sorted(usable, key=lambda x: (-round(x["vram_gb"]), x["index"]))]
else:
try:
sel = [int(x) for x in str(text).split(",") if x.strip()]
except ValueError:
fail(f"--gpus takes AMD GPU numbers as setup lists them, e.g. --gpus 1,0 (or --gpus all), not {text!r}")
if len(sel) < 2 or len(set(sel)) != len(sel):
fail("--gpus takes two or more different GPUs, e.g. --gpus 1,0 (one GPU: --gpu 1)",
"this PC has " + (f"{len(usable)} AMD card{'s' if len(usable) != 1 else ''} Strata can use"
+ (": " + ", ".join(f"GPU {g['index']} ({g['name']})" for g in usable) if usable else "")))
byid = {g["index"]: g for g in amd}
for i in sel:
g = byid.get(i)
p = "not found on this PC" if g is None else amd_problem(g)
if p is not None:
fail(f"AMD GPU {i}{'' if g is None else ' (' + g['name'] + ')'} cannot be used: {p}",
("use these together: --gpus " + ",".join(str(x["index"]) for x in usable)) if len(usable) >= 2 else
("use one card: --gpu " + str(usable[0]["index"])) if usable else f"the AMD backend runs on {AMD_CARDS}")
return [byid[i] for i in sel]
# ------------------------------------------------------------------------------------------------ AMD on Windows
# Windows has no KFD topology: the cards are found from the display adapters (Win32_VideoController: the ones present,
# with their PCI ids) and the display-class registry (each adapter's 64-bit VRAM size), before any AMD software is
# needed. The engine is the ready-made HIP one (WIN_HIP_ASSET: strata.exe, strata-device.exe and the ROCm libraries it
# loads, built by tools/hip/build_windows.bat); it needs only the AMD driver. Once it is installed, the cards are
# numbered as the HIP runtime numbers them (`strata-device --list-devices`): an integrated Radeon takes HIP's device 0
# and pushes the discrete card to 1, which the display-adapter order does not show (#325).
WIN_HIP_ASSET = "strata-windows-x64-hip.zip"
WIN_HIP_MIN_ENGINE = max(MIN_ENGINE, (0, 1, 33)) # the first release with a Windows HIP engine
WIN_AMD_DRIVER = "https://www.amd.com/en/support/download/drivers.html"
# PCI device ids (VEN_1002) of the cards the AMD backend knows; the names below cover a card whose id is not listed
_WIN_AMD_DID = {0x744C: "gfx1100", 0x7448: "gfx1100", 0x745E: "gfx1100", # RX 7900 XTX/XT/GRE, W7900, W7800
0x747E: "gfx1101", # RX 7800 XT / 7700 XT
0x7480: "gfx1102", # RX 7600 / 7600 XT
0x7590: "gfx1200", # RX 9060 XT
0x7550: "gfx1201", 0x7551: "gfx1201", # RX 9070 / 9070 XT, AI PRO R9700
0x73BF: "gfx1030", 0x73AF: "gfx1030", 0x73A5: "gfx1030"} # RX 6800 / 6800 XT / 6900 XT / 6950 XT
_WIN_AMD_NAME = ((re.compile(r"\b9070\b|R9700", re.I), "gfx1201"),
(re.compile(r"\b9060\b", re.I), "gfx1200"),
(re.compile(r"RX\s*7900|W7900|W7800", re.I), "gfx1100"),
(re.compile(r"RX\s*7800|RX\s*7700(?!\s*S)|W7700", re.I), "gfx1101"),
(re.compile(r"RX\s*7600|W7600|W7500", re.I), "gfx1102"),
(re.compile(r"RX\s*6800(?!\s*[MS])|RX\s*6900|RX\s*6950|W6800", re.I), "gfx1030"))
_DISPLAY_CLASS = r"SYSTEM\CurrentControlSet\Control\Class\{4d36e968-e325-11ce-bfc1-08002be10318}"
def _win_display_adapters() -> list[dict]:
"""The display adapters present ({"name", "pnp"}), from Win32_VideoController."""
ps = ("Get-CimInstance Win32_VideoController | ForEach-Object { $_.Name + '|' + $_.PNPDeviceID + '|' + "
"$_.AdapterRAM }")
text = out(["powershell", "-NoProfile", "-NonInteractive", "-Command", ps])
found = []
for line in text.splitlines():
parts = line.strip().split("|")
if len(parts) >= 2 and parts[1]:
ram = parts[2] if len(parts) > 2 else ""
found.append({"name": parts[0].strip(), "pnp": parts[1].strip(),
"ram": int(ram) if ram.strip().isdigit() else 0})
return found
def _win_display_registry() -> list[dict]:
"""Each display driver instance's description, matching PCI id and VRAM size (the 64-bit value; the WMI one stops
at 4 GB), from the display-class registry key - readable without admin."""
found = []
try:
import winreg
cls = winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, _DISPLAY_CLASS)
except (ImportError, OSError):
return found
i = 0
while True:
try:
sub = winreg.EnumKey(cls, i)
except OSError:
break
i += 1
if not sub.isdigit():
continue
try:
key = winreg.OpenKey(cls, sub)
except OSError:
continue
vals = {}
for name in ("DriverDesc", "MatchingDeviceId", "HardwareInformation.qwMemorySize", "DriverVersion"):
try:
vals[name] = winreg.QueryValueEx(key, name)[0]
except OSError:
pass
found.append(vals)
return found
def _pci_device_id(text: str) -> int | None:
m = re.search(r"VEN_1002&DEV_([0-9A-F]{4})", str(text or ""), re.I)
return int(m.group(1), 16) if m else None
def win_amd_arch(device_id: int | None, name: str) -> str:
"""A Windows AMD adapter's architecture from its PCI device id, else its name; "" when it is none Strata knows
(an integrated Radeon, an older card)."""
if device_id in _WIN_AMD_DID:
return _WIN_AMD_DID[device_id]
return next((a for rx, a in _WIN_AMD_NAME if rx.search(name or "")), "")
def amd_gpus_windows(adapters=None, registry=None) -> list[dict]:
"""The AMD display adapters present, in display-adapter order (setup's numbering until the HIP engine is
installed), with the arch Strata would run them as ("" = unknown: listed, not supported). adapters / registry:
tools/test_setup_amd.py passes mocked ones."""
adapters = _win_display_adapters() if adapters is None else adapters
registry = _win_display_registry() if registry is None else registry
if not adapters: # no WMI answer: the registry alone (it can list removed cards)
adapters = [{"name": r.get("DriverDesc", ""), "pnp": r.get("MatchingDeviceId", ""), "ram": 0} for r in registry]
used = set()
found = []
for ad in adapters:
did = _pci_device_id(ad.get("pnp"))
if did is None:
continue # not an AMD (VEN_1002) PCI device
vram, driver = 0.0, ""
for k, r in enumerate(registry): # the same card's driver instance: its 64-bit VRAM size
if k in used or _pci_device_id(r.get("MatchingDeviceId")) != did:
continue
if r.get("DriverDesc") and ad.get("name") and r["DriverDesc"].strip() != ad["name"].strip():
continue
used.add(k)
mem = r.get("HardwareInformation.qwMemorySize")
if isinstance(mem, bytes):
mem = int.from_bytes(mem[:8], "little")
vram = int(mem) / 2 ** 30 if isinstance(mem, int) and mem > 0 else 0.0
driver = str(r.get("DriverVersion") or "")
break
if vram == 0.0 and ad.get("ram", 0) > 0:
vram = ad["ram"] / 2 ** 30 # WMI's 32-bit figure (at most 4 GB)
arch = win_amd_arch(did, ad.get("name", ""))
name = ad.get("name") or AMD_NAMES.get(arch, f"AMD Radeon (device {did:04X})")
found.append({"index": len(found), "name": name, "vram_gb": vram, "arch": arch or f"unknown (PCI {did:04X})",
"driver": driver or "amd", "vendor": "amd"})
return found
def hip_devices(probe: Path | None = None, text: str | None = None) -> list[dict] | None:
"""The GPUs the HIP runtime enumerates, numbered as HIP_VISIBLE_DEVICES numbers them, from the installed engine's
`strata-device --list-devices`; None when there is no HIP engine here or it does not answer. text: its output
(tests)."""
if text is None:
probe = probe or ROOT / "engine" / ("strata-device.exe" if WIN else "strata-device")
try:
hip_engine = json.loads((probe.parent / "BUILD.json").read_text()).get("backend") == "hip"
except (OSError, ValueError):
hip_engine = False
if not probe.exists() or not hip_engine:
return None
try:
hip_runtime_beside_exe(probe.parent) # #468 #461: not the driver's System32 copy
env = dict(os.environ) # the ready-made engine's ROCm DLLs (rocm/bin beside it)
env["PATH"] = os.pathsep.join([str(d) for d in hip_lib_dirs(probe.parent)] + [env.get("PATH", "")])
r = subprocess.run([str(probe), "--list-devices"], capture_output=True, text=True, timeout=120,
cwd=str(probe.parent), env=env)
except (OSError, subprocess.TimeoutExpired):
return None
if r.returncode != 0:
return None
text = r.stdout
found = []
for line in text.splitlines():
m = re.match(r"device\s+(\d+):\s*(.*)$", line.strip())
if m:
found.append({"index": int(m.group(1)), "name": m.group(2).strip(), "vram_gb": 0.0, "arch": "",
"driver": "hip", "vendor": "amd"})
continue
if not found:
continue
a = re.match(r"arch\s+(gfx[0-9a-f]+)\s*,\s*([\d.]+)\s*GiB", line.strip())
if a and not found[-1]["arch"]:
found[-1]["arch"], found[-1]["vram_gb"] = a.group(1), float(a.group(2))
elif line.strip().startswith("cannot run:"):
found[-1]["cannot_run"] = line.strip()[len("cannot run:"):].strip()
for g in found:
if not g["arch"]:
g["arch"] = "unknown"
if g["arch"] in AMD_NAMES and g["name"] in ("", "AMD Radeon Graphics"):
g["name"] = AMD_NAMES[g["arch"]]
return found
def hip_lib_dirs(eng: Path) -> list[Path]:
"""Where the ready-made Windows HIP engine's ROCm DLLs are (its BUILD.json "lib_dirs", relative to engine/)."""
try:
rel = json.loads((eng / "BUILD.json").read_text()).get("lib_dirs") or []
except (OSError, ValueError):
rel = []
return [eng / d for d in rel if (eng / d).is_dir()]
# #468 #461: the HIP runtime the ready-made engine was built with, next to strata.exe. Windows looks for an imported
# DLL in the exe's folder, then System32, and only then on PATH (where rocm/bin is): an AMD driver that installs its own
# amdhip64_7.dll in System32 won, and the bundled rocBLAS/hipBLAS ran on that runtime - an access violation (W7900) or
# hipErrorInvalidDeviceFunction (7900 XTX) on the first prompt. Only the runtime and the compiler it loads by name:
# rocBLAS/hipBLASLt stay in rocm/bin, where they find their kernel libraries and ../.kpack.
HIP_RUNTIME_DLLS = ("amdhip64_*.dll", "amd_comgr*.dll")
def hip_runtime_beside_exe(eng: Path) -> None:
"""Copy the bundled HIP runtime DLLs from rocm/bin next to the engine's exes when missing or different (a 0.1.34
install, whose zip had them in rocm/bin only, is fixed on its next start)."""
for d in hip_lib_dirs(eng):
for pat in HIP_RUNTIME_DLLS:
for src in d.glob(pat):
dst = eng / src.name
try:
if dst.exists() and dst.stat().st_size == src.stat().st_size and \
dst.stat().st_mtime >= src.stat().st_mtime:
continue
shutil.copy2(src, dst)
except OSError as e: # e.g. the engine is running and holds the old copy
warn(f"could not put {src.name} next to the AMD engine ({e}); if the engine stops on its first "
"request, close Strata and run START-HERE.bat again")
def hip_match(card: dict, listed: list[dict], hip: list[dict]) -> dict | None:
"""The HIP device that is setup's `card` (from `listed`, the display-adapter order): the k-th device of the same
architecture, k = the card's rank among the listed cards of that architecture. None when HIP has no such card."""
same = [g["index"] for g in listed if g["arch"] == card["arch"]]
k = same.index(card["index"]) if card["index"] in same else 0
cand = [g for g in hip if g["arch"] == card["arch"]]
return cand[k] if k < len(cand) else None
def hip_card(eng: Path, gpu: dict, listed: list[dict]) -> dict:
"""Windows: setup's chosen AMD card as the installed HIP engine numbers it (HIP_VISIBLE_DEVICES), checked by the
engine itself before the model download: an integrated Radeon is HIP's device 0 (#325), and a PC without a
working AMD driver stops here with what to install."""
hip = hip_devices(eng / "strata-device.exe")
hint = (f"install or update the AMD driver (AMD Software: Adrenalin Edition) from {WIN_AMD_DRIVER}, restart the "
f"PC and run this again; {eng / 'strata-device.exe'} --list-devices shows what the HIP runtime sees")
if not hip:
fail("the AMD HIP runtime finds no GPU (the ready-made engine's device check)", hint)
m = hip_match(gpu, listed, hip) if gpu.get("driver") != "hip" else \
next((x for x in hip if x["index"] == gpu["index"]), None)
if m is None:
fail(f"the HIP runtime does not list your {gpu['name']} ({gpu['arch']})", hint)
if amd_problem(m):
fail(f"HIP device {m['index']} ({m['name']}) cannot be used: {amd_problem(m)}")
if gpu.get("driver") != "hip" and m["index"] != gpu["index"]:
ok(f"HIP numbers this card {m['index']} (an integrated GPU comes first): the engine is pointed at it")
return {**gpu, "index": m["index"], "count": len(hip), "vram_gb": m["vram_gb"] or gpu["vram_gb"], "driver": "hip"}
def get_prebuilt_hip(url_base, gpu, updating=False) -> Path | None:
"""The ready-made Windows HIP engine (WIN_HIP_ASSET) in engine/, kept between runs; None when it cannot be had
(not published for this version, no internet) or has no code for the card."""
eng = ROOT / "engine"
info = eng / "BUILD.json"
if info.exists() and (eng / EXE).exists():
try:
meta = json.loads(info.read_text())
except ValueError:
meta = {}
ver = tuple(int(x) for x in str(meta.get("version", "0")).split(".")[:3] if x.isdigit())
if meta.get("backend") == "hip" and meta.get("source") == "prebuilt" and ver >= WIN_HIP_MIN_ENGINE and \
gpu["arch"] in meta.get("archs", []) and not updating:
ok("ready-made AMD engine already installed")
return eng
if not url_base:
return None
eng.mkdir(exist_ok=True)
z = eng / WIN_HIP_ASSET
bases = prebuilt_bases(url_base)
for i, base in enumerate(bases):
if not base.startswith(("http://", "https://")):
break
try:
req = urllib.request.Request(base + WIN_HIP_ASSET, method="HEAD", headers={"User-Agent": "strata-setup"})
urllib.request.urlopen(req, timeout=60).close()
break
except OSError as e:
if i + 1 < len(bases):
say(f" No ready-made AMD engine for v{source_version()} ({e}): the latest release instead")
continue
warn(f"no ready-made AMD engine at {base} ({e})")
return None
say(" Downloading the ready-made Strata engine for AMD GPUs (with the ROCm libraries it uses) ...")
download(base + WIN_HIP_ASSET, z, "Strata AMD engine")
tmp = eng / "_unpack"
shutil.rmtree(tmp, ignore_errors=True)
with zipfile.ZipFile(z) as f:
f.extractall(tmp)
try:
meta = json.loads((tmp / "BUILD.json").read_text())
except (OSError, ValueError):
meta = {}
ver = tuple(int(x) for x in str(meta.get("version", "0")).split(".")[:3] if x.isdigit())
why = None
if meta.get("backend") != "hip" or not (tmp / EXE).exists():
why = "it is not a HIP engine"
elif ver < WIN_HIP_MIN_ENGINE:
why = f"it is version {meta.get('version')}; this setup needs {'.'.join(map(str, WIN_HIP_MIN_ENGINE))}"
elif gpu["arch"] not in meta.get("archs", []):
why = f"it is built for {', '.join(meta.get('archs', []))}; your GPU is {gpu['arch']}"
if why:
warn(f"the ready-made AMD engine at {base} cannot be used: {why}")
shutil.rmtree(tmp, ignore_errors=True)
drop_archive(z)
return None
for p in tmp.iterdir():
dst = eng / p.name
if dst.exists():
shutil.rmtree(dst) if dst.is_dir() else dst.unlink()
p.replace(dst)
shutil.rmtree(tmp, ignore_errors=True)
drop_archive(z)
hip_runtime_beside_exe(eng) # #468 #461
ok(f"ready-made AMD engine {meta.get('version', '')} for {', '.join(meta.get('archs', []))} "
f"(ROCm {meta.get('rocm', '?')})")
return eng
def rocm_version(root):
"""(major, minor) of a ROCm install, from rocm-core's header; None when it has none."""
try:
text = (Path(root) / "include" / "rocm-core" / "rocm_version.h").read_text()
return tuple(int(re.search(rf"#define\s+ROCM_VERSION_{k}\s+(\d+)", text).group(1)) for k in ("MAJOR", "MINOR"))
except (OSError, AttributeError, ValueError):
return None
def rocm_dev_missing(sysroot: Path) -> list:
"""#446: the HIP development files the engine's build needs that a system ROCm lacks (a runtime-only install has
hipcc and libhipblas but not these, and cmake's enable_language(HIP) then fails on the hip-lang package)."""
lang = "cmake/hip-lang/hip-lang-config.cmake" # where CMake's HIP support looks for it
need = {"lib/" + lang: [sysroot / d / lang for d in ("lib", "lib64", "lib/x86_64-unknown-linux-gnu")],
"include/hip/hip_runtime.h": [sysroot / "include" / "hip" / "hip_runtime.h"]}
return [name for name, paths in need.items() if not any(p.is_file() for p in paths)]
def rocm_root(archs):
"""ROCm for compiling and running the HIP engine for `archs` (one arch or a list: the cards of a layer split):
(root, library folders). A system ROCm 7 with hipcc, hipBLAS and the HIP development files (#446), else AMD's
TheRock wheels (ROCM_VERSION, from the card family's index) installed into .venv."""
archs = [archs] if isinstance(archs, str) else list(archs)
sysroot = Path(os.environ.get("ROCM_PATH") or "/opt/rocm")
if (sysroot / "bin" / "hipcc").exists() and list((sysroot / "lib").glob("libhipblas.so*")):
ver = rocm_version(sysroot)
missing = rocm_dev_missing(sysroot)
if (ver is None or ver >= ROCM_SYSTEM_MIN) and not missing:
return sysroot, [str(sysroot / "lib")]
if ver is not None and ver < ROCM_SYSTEM_MIN:
warn(f"the ROCm in {sysroot} is {ver[0]}.{ver[1]}; Strata needs {ROCM_SYSTEM_MIN[0]}.{ROCM_SYSTEM_MIN[1]} "
"or newer: using AMD's wheels in .venv instead")
else: # #446: a runtime-only ROCm (no -dev packages): cmake would fail
warn(f"the ROCm in {sysroot} has no HIP development files ({', '.join(missing)}): using AMD's wheels in "
".venv instead (or install them, e.g. AMD's amdrocm-core-dev package for your ROCm and card)")
indexes = list(dict.fromkeys(rocm_index(a) for a in archs))
if len(indexes) > 1: # TheRock's wheels hold one GPU family's libraries
fail(f"cards of two GPU families ({', '.join(archs)}) need a system ROCm 7 (in /opt/rocm): AMD's Python "
"wheels come per family", "install ROCm 7 system-wide, or use cards of one family (--gpu N for one card)")
index = indexes[0]
stamp = Path(sys.prefix) / ".strata-rocm.json"
have = json.loads(stamp.read_text()) if stamp.exists() else {}
if have.get("version") != ROCM_VERSION or have.get("index") != index:
say(f" Installing ROCm {ROCM_VERSION} for AMD GPUs into .venv (AMD's TheRock wheels, ~10 GB, no sudo) ...")
pip = [sys.executable, "-m", "pip", "install", "--quiet", "--disable-pip-version-check", "--index-url", index]
if have.get("version") == ROCM_VERSION: # the same version for another GPU family: its own libraries
run(pip + ["--force-reinstall", "--no-deps", f"rocm=={ROCM_VERSION}"])
run(pip + [f"rocm[libraries,devel]=={ROCM_VERSION}"])
stamp.write_text(json.dumps({"version": ROCM_VERSION, "index": index}))
sdk = Path(sys.executable).parent / "rocm-sdk"
root = Path(out([str(sdk), "path", "--root"]).strip())
if not (root / "llvm" / "bin" / "clang++").exists():
fail(f"ROCm was installed but its compiler is missing ({root})",
f"remove {stamp} and run this again; or install ROCm 7 system-wide")
# the card family's libraries only (gfx120X-all -> _rocm_sdk_libraries_gfx120X_all): another family's
# libhipblaslt.so first on the path would have no kernels for this card
family = "_rocm_sdk_libraries_" + index.rstrip("/").rsplit("/", 1)[-1].replace("-", "_")
dirs = [str(root / "lib")]
for sp in {Path(p) for p in sys.path if p.endswith("site-packages")}:
libs = sorted(sp.glob("_rocm_sdk_libraries_*"))
libs = [d for d in libs if d.name.lower() == family.lower()] or libs
dirs += [str(d / "lib") for d in libs if (d / "lib").is_dir()]
ok(f"ROCm: {root}")
return root, list(dict.fromkeys(dirs))
def hipblaslt_version(lib_dirs):
"""The installed hipBLASLt's version as the engine reads it (hipblasLtGetVersion: 1.4.1 -> 100401), from the
header of the ROCm whose libraries the engine loads; None when not found."""
for d in lib_dirs:
try:
text = (Path(d).parent / "include" / "hipblaslt" / "hipblaslt-version.h").read_text()
v = [int(re.search(rf"#define\s+HIPBLASLT_VERSION_{k}\s+(\d+)", text).group(1))
for k in ("MAJOR", "MINOR", "PATCH")]
except (OSError, AttributeError, ValueError):
continue
return v[0] * 100000 + v[1] * 100 + v[2]
return None
def hipblaslt_table(arch, lib_dirs, ver=None):
"""tools/hip/<arch>-hipblaslt-<version>.txt for this card AND the installed hipBLASLt, else None: its solution
ids are valid only for that pair (the engine refuses any other table and uses plain hipBLAS). ver: the
hipBLASLt version the ready-made Windows engine ships (its BUILD.json), else read from the installed headers."""
ver = ver or hipblaslt_version(lib_dirs)
table = ROOT / "tools" / "hip" / f"{arch}-hipblaslt-{ver}.txt"
if ver is not None and table.exists():
head = table.read_text().split("\n", 2)[:2]
if f"STRATA_HIPBLASLT_TUNING_V1 {arch} {ver}" in (h.strip() for h in head):
ok(f"hipBLASLt tuning table: {table.name} (faster prompts)")
return table
have = sorted(p.name for p in (ROOT / "tools" / "hip").glob(f"{arch}-hipblaslt-*.txt"))
warn(f"no hipBLASLt tuning table for {arch} with hipBLASLt {ver or '(version unknown)'}"
+ (f" (have: {', '.join(have)})" if have else "")
+ ": the prompt's dense matrix products use plain hipBLAS (tools/hip/tune_hipblaslt makes a table: "
"docs/AMD_HIP.md, Tuning table)")
return None
def build_engine_hip(gpu, llama, vision="none") -> Path:
"""Compile the HIP engine for this AMD GPU into engine/ (again only when its source changed: a `git pull`).
gpu["archs"]: every architecture it needs code for (the cards of a layer split), else gpu["arch"]. vision "cpu"
(#304): the image encoder too, for the CPU (there is no HIP encoder build yet)."""
eng = ROOT / "engine"
eng.mkdir(exist_ok=True)
stamp = eng / "BUILD.json"
meta = json.loads(stamp.read_text()) if stamp.exists() else {}
src, vsrc = source_hash(ENGINE_SOURCES), source_hash(VISION_SOURCES)
archs = sorted(set(gpu.get("archs") or [gpu["arch"]]))
has_archs = set(archs) <= set(meta.get("archs", []))
engine_ok = meta.get("backend") == "hip" and (eng / EXE).exists() and meta.get("src") == src and has_archs
vision_ok = vision == "none" or ((eng / VEXE).exists() and meta.get("vision_src") == vsrc)
if engine_ok and vision_ok:
ok("engine already built for this PC")
return eng
if engine_ok:
return build_vision_cpu(eng, stamp, meta, llama, vsrc)
if not (shutil.which("c++") or shutil.which("g++")) or not shutil.which("git"):
fail("a C++ compiler and git are needed to compile the AMD engine",
"Ubuntu/Debian: sudo apt install build-essential git Fedora: sudo dnf install gcc-c++ git")
root, dirs = rocm_root(archs)
libs = [str(Path(d).parent) for d in dirs[1:]]
bitcode = next((p for p in (root / "lib" / "llvm" / "amdgcn" / "bitcode", root / "amdgcn" / "bitcode") if p.is_dir()),
root / "amdgcn" / "bitcode")
os.environ.update({"HIP_PLATFORM": "amd", "HIP_COMPILER": "clang", "HIP_RUNTIME": "rocclr", "ROCM_PATH": str(root),
"HIP_PATH": str(root)})
os.environ["LD_LIBRARY_PATH"] = os.pathsep.join(dirs + [os.environ.get("LD_LIBRARY_PATH", "")]).rstrip(os.pathsep)
os.environ["PATH"] = os.pathsep.join([str(root / "bin"), str(root / "llvm" / "bin"), os.environ.get("PATH", "")])
say(" The engine's source changed: compiling it again (only what changed, a few minutes) ..."
if meta.get("backend") == "hip" and (eng / EXE).exists() and has_archs
else f" Compiling the Strata engine for your AMD GPU{'s' if len(archs) > 1 else ''} ({', '.join(archs)}; "
"10-20 minutes, once) ...")
cmake_build(ROOT, ROOT / "build-hip", "strata",
["-DSTRATA_ENABLE_HIP=ON", "-DSTRATA_ENABLE_CUDA=OFF", "-DSTRATA_BUILD_TESTS=OFF",
"-DSTRATA_PREFILL_MMQ=ON", "-DCMAKE_HIP_ARCHITECTURES=" + ";".join(archs),
f"-DCMAKE_HIP_COMPILER={root / 'llvm' / 'bin' / 'clang++'}", f"-DCMAKE_HIP_COMPILER_ROCM_ROOT={root}",
"-DCMAKE_PREFIX_PATH=" + ";".join([str(root), *libs]),
f"-DCMAKE_HIP_FLAGS=--rocm-path={root} --rocm-device-lib-path={bitcode}",
f"-DSTRATA_GGML_DIR={llama}"], None, "")
shutil.copy2(ROOT / "build-hip" / EXE, eng / EXE)
meta = {"source": "local-hip", "backend": "hip", "version": source_version(), "archs": archs, "vision": "none",
"lib_dirs": dirs, "src": src}
if vision != "none":
return build_vision_cpu(eng, stamp, meta, llama, vsrc)
stamp.write_text(json.dumps(meta, indent=1))
ok(f"engine compiled: {eng / EXE}")
return eng
def hip_vision(asked) -> str:
"""The image encoder with the AMD backend (--vision): the CPU one when asked for (#304); a HIP (GPU) encoder build
is a later step, so `yes`/`gpu` leave images off, as before, and say how to get them."""
if asked in ("yes", "gpu"):
warn("the AMD backend has no GPU image encoder yet: images off"
+ ("" if WIN else " (--vision cpu reads them on the CPU)"))
if asked == "cpu" and WIN:
warn("images on the CPU with an AMD card are Linux-only for now (the ready-made Windows AMD engine has no "
"image encoder): images off")
return "none"
return "cpu" if asked == "cpu" else "none"
def build_vision_cpu(eng: Path, stamp: Path, meta: dict, llama, vsrc) -> Path:
"""#304: the CPU image encoder beside the HIP engine (tools/vision without CUDA), recorded in its BUILD.json."""
if not ((eng / VEXE).exists() and meta.get("vision_src") == vsrc):
say(" Compiling the image encoder (for the CPU) ...")
cmake_build(ROOT / "tools" / "vision", ROOT / "build-vision", "strata-vision",
[f"-DLLAMA_DIR={llama}", "-DSTRATA_VISION_CUDA=OFF"], None, "")
shutil.copy2(ROOT / "build-vision" / "bin" / VEXE, eng / VEXE)
stamp.write_text(json.dumps({**meta, "vision": "cpu", "vision_src": vsrc}, indent=1))
ok(f"engine: {eng / EXE}, image encoder (CPU): {eng / VEXE}")
return eng
# ------------------------------------------------------------------------------------------------ the engine
def driver_major(gpu):
try:
return int(gpu["driver"].split(".")[0])
except (ValueError, KeyError):
return 0
def prebuilt_bases(url_base) -> list[str]:
"""Where to look for the ready-made engine, in order (each ending in a slash). The default: the release of this
checkout's version first, then the latest (#214); an explicit --prebuilt / STRATA_PREBUILT_URL: only that."""
base = url_base if url_base.endswith(("/", "\\")) else url_base + "/"
if base != PREBUILT_URL:
return [base]
return [PREBUILT_TAG_URL.format(version=source_version()), base]
def get_prebuilt(url_base, gpu, vision, updating=False) -> Path | None:
"""The ready-made engine in engine/ (kept between runs), or None when there is none for this PC.
updating: called to replace an installed engine, which starts instead when this fails (no compile)."""
eng = ROOT / "engine"
info = eng / "BUILD.json"
if info.exists() and (eng / EXE).exists() and json.loads(info.read_text()).get("backend") != "hip":
meta = json.loads(info.read_text())
ver = tuple(int(x) for x in str(meta.get("version", "0")).split(".")[:3] if x.isdigit())
if meta.get("source") == "local": # compiled here: build_engine checks its source and cards
return None
have = [int(a) for a in meta.get("archs", [])]
miss = [int(x) for x in gpu.get("archs", [gpu["arch"]])
if have and int(x) not in have and not (meta.get("ptx") and int(x) > max(have))]
if miss: # a card it has no code for (#128): compiled here instead
warn(f"the installed engine is built for {', '.join(str(a) for a in have)}; your GPU is "
f"{', '.join(str(x) for x in miss)}: compiling instead")
return None
if ver >= MIN_ENGINE:
ok("ready-made engine already installed")
return eng
say(f" Updating the ready-made engine ({meta.get('version')} -> {'.'.join(map(str, MIN_ENGINE))} or newer) ...")
info.unlink()
if not url_base:
return None
z = ROOT / "engine" / PREBUILT_ASSET
bases = prebuilt_bases(url_base)
for i, base in enumerate(bases):
if not base.startswith(("http://", "https://")):
break
try: # not published (yet), or no internet: compile instead
req = urllib.request.Request(base + PREBUILT_ASSET, method="HEAD", headers={"User-Agent": "strata-setup"})
urllib.request.urlopen(req, timeout=60).close()
break
except OSError as e:
if i + 1 < len(bases): # #214: this checkout's release is not published (yet)
say(f" No ready-made engine for v{source_version()} ({e}): the latest release instead")
continue
warn(f"no ready-made engine at {base} ({e})" + ("" if updating else ": compiling instead"))
return None
say(" Downloading the ready-made Strata engine ...")
download(base + PREBUILT_ASSET, z, "Strata engine")
tmp = ROOT / "engine" / "_unpack"
shutil.rmtree(tmp, ignore_errors=True)
with zipfile.ZipFile(z) as f:
f.extractall(tmp)
meta = json.loads((tmp / "BUILD.json").read_text())
if tuple(int(x) for x in str(meta.get("version", "0")).split(".")[:3] if x.isdigit()) < MIN_ENGINE:
need = ".".join(map(str, MIN_ENGINE))
if updating: # these files are newer than the published release (#58)
warn(f"engine {need} is not published yet (the release may still be uploading): run this again "
f"in a few minutes to update it")
else:
warn(f"the ready-made engine at {base} is version {meta.get('version')}; this setup needs "
f"{need}: compiling instead")
shutil.rmtree(tmp, ignore_errors=True)
drop_archive(z)
return None
archs = [int(a) for a in meta.get("archs", [])]
miss = [int(x) for x in gpu.get("archs", [gpu["arch"]])
if int(x) not in archs and not (meta.get("ptx") and int(x) > max(archs))]
if miss:
warn(f"the ready-made engine is built for {', '.join(str(a) for a in archs)}; your GPU is "
f"{', '.join(str(x) for x in miss)}" + ("" if updating else ": compiling instead"))
shutil.rmtree(tmp, ignore_errors=True)
drop_archive(z)
return None
for p in tmp.iterdir():
dst = eng / p.name
if dst.exists():
shutil.rmtree(dst) if dst.is_dir() else dst.unlink()
p.replace(dst)
shutil.rmtree(tmp, ignore_errors=True)
drop_archive(z)
if not (eng / EXE).exists():
fail("the ready-made engine archive has no " + EXE)
if not WIN:
for x in (EXE, VEXE):
if (eng / x).exists():
(eng / x).chmod(0o755)
ok(f"ready-made engine {meta.get('version', '')} for {', '.join('sm_' + str(a) for a in archs)} (CUDA "
f"{meta.get('cuda', '?')})")
return eng
def update_installed_engine(url_base) -> None:
"""An installed ready-made engine older than MIN_ENGINE is replaced before the model starts, so a plain
START-HERE.bat on an existing install picks up a new release. If that cannot happen (no internet, the model
still running, no ready-made engine for this GPU) the installed engine is kept and starts as before."""
eng = ROOT / "engine"
info = eng / "BUILD.json"
if not info.exists() or not (eng / EXE).exists():
return
meta_text = info.read_text()
meta = json.loads(meta_text)
if meta.get("backend") == "hip" and WIN: # AMD on Windows: the ready-made HIP engine, when older
ver = tuple(int(x) for x in str(meta.get("version", "0")).split(".")[:3] if x.isdigit())
if meta.get("source") == "prebuilt" and ver < WIN_HIP_MIN_ENGINE:
try:
g = next((x for x in amd_gpus() if amd_problem(x) is None), None)
if g is None:
raise RuntimeError("no supported AMD GPU found")
say(f" Updating the ready-made AMD engine ({meta.get('version')} -> "
f"{'.'.join(map(str, WIN_HIP_MIN_ENGINE))} or newer) ...")
if get_prebuilt_hip(url_base, g, updating=True) is None:
raise RuntimeError("not published yet")
except (Exception, SystemExit) as e:
warn(f"could not update the AMD engine{'' if isinstance(e, SystemExit) else f' ({e})'}: "
"starting the installed one")
return
if meta.get("backend") == "hip": # AMD: compiled here, again when its source changed
if meta.get("src") != source_hash(ENGINE_SOURCES):
try:
usable = [x for x in amd_gpus() if amd_problem(x) is None]
g = next((x for x in usable if x["arch"] in meta.get("archs", [])), usable[0] if usable else None)
if g is None:
raise RuntimeError("no supported AMD GPU found")
# every architecture it was built for (a layer split across two families keeps both)
build_engine_hip({**g, "archs": [x for x in meta.get("archs", []) if x in AMD_ARCHS] or [g["arch"]]},
get_llama_cpp(), meta.get("vision") or "none")
except (Exception, SystemExit) as e:
warn(f"could not compile the updated engine{'' if isinstance(e, SystemExit) else f' ({e})'}: "
"starting the installed one")
return
ver = tuple(int(x) for x in str(meta.get("version", "0")).split(".")[:3] if x.isdigit())
local = meta.get("source") == "local"
vision = meta.get("vision") or "none"
if local: # compiled here: is it older than the source (a git pull)?
if meta.get("src") == source_hash(ENGINE_SOURCES) and \
(vision == "none" or meta.get("vision_src") == source_hash(VISION_SOURCES)):
return
elif ver >= MIN_ENGINE:
return
try: # a running engine cannot be replaced (Windows keeps it locked)
for x in (EXE, VEXE):
if (eng / x).exists():
with open(eng / x, "r+b"):
pass
except OSError:
warn(f"engine {meta.get('version') or ''} is in use: close the model window and run this again to update it")
return
gpu = gpu_info()
if local:
try: # a failed compile must not stop the model from starting
if gpu is None:
raise RuntimeError("no NVIDIA GPU found")
gpu = {**gpu, "archs": sorted({int(gpu["arch"]), *(int(x) for x in meta.get("archs", []))})}
build_engine(gpu, vision, False, get_llama_cpp())
except (Exception, SystemExit) as e:
warn(f"could not compile the updated engine{'' if isinstance(e, SystemExit) else f' ({e})'}: starting the installed one")
return
new = None
if gpu is not None:
try:
new = get_prebuilt(url_base, gpu, "gpu", updating=True)
except Exception as e: # a failed download must not stop the model from starting
warn(f"updating the engine failed ({e})")
if new is None:
if not info.exists():
info.write_text(meta_text) # get_prebuilt drops it before downloading: put it back
warn(f"could not update the engine: starting the installed {meta.get('version')}")
return
pip_install(CUDA_WHEELS, "NVIDIA CUDA libraries (cuBLAS, CUDA runtime; ~0.4 GB)")
def install_build_tools(gpu, yes):
"""The compiler and the CUDA toolkit, installed for the user (asks once). Returns (nvcc, vcvars)."""
archs = [int(x) for x in gpu.get("archs", [gpu["arch"]])]
# #295: Pascal / Volta (STRATA_EXPERIMENTAL_SM60=1) need a CUDA 12.x toolkit - CUDA 13 cannot build sm_60/sm_70
old = min(archs) < 75
if old and max(archs) >= 120:
fail("one engine cannot be built for both an RTX 50 card (CUDA 13) and a Pascal/Volta card (CUDA 12.x)",
"choose the cards of one kind with --gpu / --gpus")
nvcc, cuda_v = find_nvcc(below=(13, 0)) if old else find_nvcc()
if old and (nvcc is None or cuda_v < (12, 0)):
fail("the experimental Pascal/Volta build (STRATA_EXPERIMENTAL_SM60=1) needs the NVIDIA CUDA Toolkit 12.x "
"(CUDA 13 cannot compile for these cards)",
"install CUDA 12.9 (or another 12.x; it can sit next to a newer one) from "
"https://developer.nvidia.com/cuda-toolkit-archive and run it again")
# RTX 50 (sm_120): CUDA 13.0 - an engine built with 12.8 crashed in the prompt path on Linux (#220)
need_cuda = (13, 0) if max(archs) >= 120 else (12, 0)
vcvars = find_vcvars() if WIN else None
have_cc = vcvars is not None if WIN else shutil.which("g++") is not None
missing = []
if not have_cc:
missing.append("Visual Studio 2022 Build Tools (C++)" if WIN else "the C++ compiler (build-essential)")
if nvcc is None or cuda_v < need_cuda:
missing.append("the NVIDIA CUDA Toolkit 13.0")
if not missing:
ok(f"build tools present (CUDA {cuda_v[0]}.{cuda_v[1]})")
return nvcc, vcvars
say(" The engine has to be compiled for your PC, which needs: " + " and ".join(missing) + ".")
say(" They can be installed now (about 8-10 GB, 15-40 minutes" + (", Windows will ask for permission" if WIN else
", sudo will ask for your password") + ").")
if ask(" Install them now?", ["y", "n"], "y", yes) != "y":
fail("the build tools are needed", "install them yourself (see README.md) and run it again")
if WIN:
if shutil.which("winget") is None:
fail("winget (Windows package manager) is not available",
"install 'App Installer' from the Microsoft Store, or install the tools by hand (README.md)")
wg = ["winget", "install", "-e", "--source", "winget", "--accept-package-agreements",
"--accept-source-agreements", "--disable-interactivity"]
if not have_cc:
run([*wg, "--id", "Microsoft.VisualStudio.2022.BuildTools", "--override",
"--quiet --wait --norestart --nocache --add Microsoft.VisualStudio.Workload.VCTools --includeRecommended"],
check=False)
if nvcc is None or cuda_v < need_cuda:
run([*wg, "--id", "Nvidia.CUDA", "--version", "13.0"], check=False)
vcvars = find_vcvars()
else:
apt = shutil.which("apt-get")
if apt is None:
fail("missing: " + " and ".join(missing) + " (the automatic install is only done on Ubuntu/Debian)",
"install them with your distribution's packages (Arch: pacman -S base-devel cuda; nvcc is found on "
"PATH, in /usr/local/cuda* and in /opt/cuda*), then run it again")
if not have_cc:
run(["sudo", "apt-get", "install", "-y", "build-essential"])
if nvcc is None or cuda_v < need_cuda:
osr = dict(line.split("=", 1) for line in open("/etc/os-release").read().splitlines() if "=" in line)
ver = osr.get("VERSION_ID", "").strip('"').replace(".", "")
if osr.get("ID") != "ubuntu" or ver not in ("2204", "2404"):
fail("the CUDA Toolkit can be installed automatically on Ubuntu 22.04 / 24.04 only",
"install it from https://developer.nvidia.com/cuda-downloads and run it again")
deb = Path("/tmp/cuda-keyring.deb")
download(f"https://developer.download.nvidia.com/compute/cuda/repos/ubuntu{ver}/x86_64/cuda-keyring_1.1-1_all.deb",
deb, "CUDA repository key")
run(["sudo", "dpkg", "-i", str(deb)])
run(["sudo", "apt-get", "update"])
run(["sudo", "apt-get", "install", "-y", "cuda-toolkit-13-0"])
nvcc, cuda_v = find_nvcc(below=(13, 0)) if old else find_nvcc()
if (WIN and find_vcvars() is None) or (not WIN and shutil.which("g++") is None):
fail("the C++ build tools did not install", "install them by hand (README.md) and run it again")
if nvcc is None or cuda_v < need_cuda:
fail("the CUDA Toolkit did not install", "install it from https://developer.nvidia.com/cuda-downloads, then run it again")
ok(f"build tools installed (CUDA {cuda_v[0]}.{cuda_v[1]})")
return nvcc, find_vcvars() if WIN else None
def cmake_build(src, bdir, target, defs, vcvars, bat_name):
cmake, ninja = find_tool("cmake"), find_tool("ninja")
if cmake is None or ninja is None:
fail("cmake / ninja not found after installing them", "run: .venv python -m pip install cmake ninja")
conf = [cmake, "-G", "Ninja", f"-DCMAKE_MAKE_PROGRAM={ninja}", "-S", str(src), "-B", str(bdir),
"-DCMAKE_BUILD_TYPE=Release", *defs]
build = [cmake, "--build", str(bdir), "--target", target, "-j", str(max(2, (os.cpu_count() or 4) // 2))]
# A failed build is tried once more: CUDA 13.0's ptxas now and then fails to parse a PTX file it just wrote, and
# the same command then gets past it (issue #45); a second attempt only compiles what is still missing.
if WIN:
bat = ROOT / bat_name
q = lambda c: " ".join(f'"{x}"' if " " in str(x) else str(x) for x in c) # noqa: E731
bat.write_text(f'@echo off\r\ncall "{vcvars}" >nul\r\n{q(conf)} || exit /b 1\r\n{q(build)} && exit /b 0\r\n'
f'echo (the build stopped - trying it once more)\r\n{q(build)} || exit /b 1\r\n',
encoding="utf-8")
run(["cmd", "/c", str(bat)])
else:
run(conf)
if run(build, check=False).returncode != 0:
say(" (the build stopped - trying it once more)")
run(build)
ENGINE_SOURCES = ("CMakeLists.txt", "src", "include", "third_party/ggml")
VISION_SOURCES = ("tools/vision",)
def source_hash(parts) -> str:
"""A fingerprint of the files a compiled engine is built from, kept in engine/BUILD.json: when a `git pull`
changes them, the engine is compiled again (issue #31)."""
h = hashlib.sha256(LLAMA_CPP_COMMIT.encode())
for part in parts:
base = ROOT / part
for f in [base] if base.is_file() else sorted(x for x in base.rglob("*") if x.is_file()):
h.update(f.relative_to(ROOT).as_posix().encode() + b"\0" + f.read_bytes().replace(b"\r\n", b"\n"))
return h.hexdigest()[:16]
def engine_defs(archs) -> list:
"""Extra CMake definitions for the engine: the experimental Pascal/Volta build (#295) for cards below sm_75."""
return ["-DSTRATA_EXPERIMENTAL_SM60=ON"] if min(int(x) for x in archs) < 75 else []
def prebuilt_vision(meta: dict, gpu: dict, vision: str) -> str:
"""The image encoder to use with a ready-made engine (`meta`: its BUILD.json). The encoder can cover fewer cards
than the engine (0.1.30/0.1.31: no RTX 20 code, #331): such a card gets the CPU encoder - the same program - instead
of compiling one, which fails on most Windows PCs (no Visual Studio / CUDA toolkit); --build compiles it."""
if vision != "gpu":
return vision
va = [int(x) for x in meta.get("vision_archs", meta.get("archs", []))]
if va and int(gpu["arch"]) not in va and not (meta.get("ptx") and int(gpu["arch"]) > max(va)):
warn(f"the ready-made image encoder has no code for your GPU (sm_{gpu['arch']}): it runs on the CPU instead "
"(images take longer; setup --build compiles one for your GPU)")
return "cpu"
return vision
def build_engine(gpu, vision, yes, llama) -> Path:
"""Compile the engine (and, for images, the encoder) for this GPU; the results go to engine/. A compiled
engine whose source files changed since (a `git pull`) is compiled again: only the changed files, a few minutes."""
eng = ROOT / "engine"
eng.mkdir(exist_ok=True)
stamp = eng / "BUILD.json"
meta = json.loads(stamp.read_text()) if stamp.exists() else {}
want_vision = vision != "none"
local = meta.get("source") == "local"
src, vsrc = source_hash(ENGINE_SOURCES), source_hash(VISION_SOURCES)
archs = sorted({int(x) for x in gpu.get("archs", [gpu["arch"]])}) # every card the model runs on
built = {int(x) for x in meta.get("archs", [])}
# a card the engine has no code for (a GPU added with --gpus, #128) needs a compile even when the source is the
# same; the compile keeps the generations it was built for
new_arch = local and not set(archs) <= built
engine_ok = local and (eng / EXE).exists() and meta.get("src") == src and not new_arch
vision_ok = not want_vision or ((eng / VEXE).exists() and (not local or meta.get("vision_src") == vsrc))
if engine_ok and vision_ok:
ok("engine already built for this PC")
return eng
if local:
archs = sorted(built | set(archs))
nvcc, vcvars = install_build_tools({**gpu, "archs": archs}, yes)
cuda_archs = ";".join(str(x) for x in archs)
if not engine_ok:
say(" Compiling the engine for " + ", ".join(f"sm_{x}" for x in archs) + " (a card it had no code for; "
"10-20 minutes, once) ..." if new_arch else
" The engine's source changed: compiling it again (only what changed, a few minutes) ..."
if local and (eng / EXE).exists() else " Compiling the Strata engine for your GPU (10-20 minutes, once) ...")
cmake_build(ROOT, ROOT / "build", "strata",
["-DSTRATA_ENABLE_CUDA=ON", "-DSTRATA_BUILD_TESTS=OFF", f"-DCMAKE_CUDA_ARCHITECTURES={cuda_archs}",
f"-DCMAKE_CUDA_COMPILER={nvcc}", f"-DSTRATA_GGML_DIR={llama}", *engine_defs(archs)],
vcvars, "build-strata.bat")
shutil.copy2(ROOT / "build" / EXE, eng / EXE)
if not vision_ok:
say(" Compiling the image encoder" + (" with CUDA (10-20 minutes, once) ..." if vision == "gpu" else " ..."))
defs = [f"-DLLAMA_DIR={llama}", f"-DSTRATA_VISION_CUDA={'ON' if vision == 'gpu' else 'OFF'}"]
if vision == "gpu":
defs += [f"-DCMAKE_CUDA_ARCHITECTURES={cuda_archs}", f"-DCMAKE_CUDA_COMPILER={nvcc}"]
cmake_build(ROOT / "tools" / "vision", ROOT / "build-vision", "strata-vision", defs, vcvars, "build-vision.bat")
shutil.copy2(ROOT / "build-vision" / "bin" / VEXE, eng / VEXE)
bindir = Path(nvcc).parent # the toolkit's own libraries (bin, bin/x64, lib64)
dirs = [str(d) for d in (bindir, bindir / "x64", bindir.parent / "lib64") if d.is_dir()]
stamp.write_text(json.dumps({"source": "local", "version": source_version(), "archs": archs,
"vision": vision,
"cuda_dirs": dirs, "src": src, "vision_src": vsrc if want_vision else None}, indent=1))
ok(f"engine compiled: {eng / EXE}")
return eng
# ------------------------------------------------------------------------------------------------ the data folder
# The model files - the GGUFs, the prepared packs and the MTP layer, 70-120 GB - live in a data folder NEXT TO the
# Strata folder (`Strata-data`), not inside it: updating Strata by unzipping a new copy used to give a new, empty
# folder and a full download again. Where it is, and which Strata folders this user ran, is kept in a small
# per-user file, so every Strata folder on the PC finds the same files.
DATA_ITEMS = ("models", "packs", "mtp")
LOW_RAM_HEADROOM_GB = 10 # RAM beside the experts: the OS, the engine's other buffers, the server
RESIDENT_ENGINE = (0, 1, 30) # the first engine with --resident-experts (the low-RAM mode's resident variant)
def low_ram_needed(model, ram) -> bool:
"""The model's experts do not fit this PC's RAM with room left for the rest: they are then mapped from the pack's
experts.bin instead of copied into RAM (the low-RAM mode)."""
return ram < MODELS[model]["arena_gb"] + LOW_RAM_HEADROOM_GB
def low_ram_gpu_gb(model, vram_gb, ctx=32768, kv="int8") -> float:
"""About how many GB of the model's experts the GPU's cache holds: its VRAM minus ~5 GB for the dense weights,
buffers and a 32K context's KV cache, minus the KV cache of a longer context (in VRAM in the low-RAM mode: its RAM
has no room for KV streaming)."""
kv_tok = 13 * (576 if kv == "q4_0" else 1056) # bytes per context token: 12 QSA layers + the draft layer
longer = max(0, ctx - 32768) * kv_tok / 1e9
return max(0.0, min(MODELS[model]["arena_gb"], vram_gb - 5 - longer))
def low_ram_gpu_share(model, vram_gb, ctx=32768, kv="int8") -> float:
"""About how much of the model's experts the GPU holds."""
return low_ram_gpu_gb(model, vram_gb, ctx, kv) / MODELS[model]["arena_gb"]
def low_ram_resident(model, ram, vram_gb, ctx=32768, kv="int8") -> bool:
"""In the low-RAM mode: the experts the GPU does not hold fit the RAM with the usual room beside them, so they are
copied into RAM once (the resident variant, `--resident-experts`) instead of being read through the OS file cache
(plain `--mmap-experts`, which a PC this short of RAM keeps re-reading from the SSD)."""
rest = MODELS[model]["arena_gb"] - low_ram_gpu_gb(model, vram_gb, ctx, kv)
return ram >= rest + LOW_RAM_HEADROOM_GB
def low_ram_fits(model, ram, vram_gb) -> bool:
"""In the low-RAM mode: the experts the GPU does not hold fit the RAM left beside the rest (as file cache)."""
arena = MODELS[model]["arena_gb"]
return ram - 6 + max(0.0, vram_gb - 5) >= arena
def low_ram_one_gpu_why(model, ram, choice, sel=None) -> list[str]:
"""#250: why the low-RAM mode recommends one GPU, with the RAM math that turned it on; #364 #384: and how to use
all of them (sel: the cards, for the --gpus example)."""
arena, need = MODELS[model]["arena_gb"], MODELS[model]["arena_gb"] + LOW_RAM_HEADROOM_GB
if choice == "auto":
why = [f"Why: {model}'s experts are {arena:.0f} GB and must fit in RAM with ~{LOW_RAM_HEADROOM_GB} GB beside "
f"them for the OS and the rest: {arena:.0f} + {LOW_RAM_HEADROOM_GB} = {need:.0f} GB, and this PC has "
f"{ram:.0f} GB.",
"So setup uses the low-RAM mode: the experts come from the model's file (copied into RAM as far as "
"it fits), the GPU holds the most-used ones."]
else:
why = [f"Why: you chose the low-RAM mode (--low-ram {choice}); without it {model} needs {arena:.0f} + "
f"{LOW_RAM_HEADROOM_GB} = {need:.0f} GB of RAM, this PC has {ram:.0f} GB."]
return why + ["Its resident variant (the experts the GPU does not hold copied into RAM once: steady RAM use) runs "
"on one GPU: the engine has no layer split for it yet.",
f"To use all the GPUs: --gpus {','.join(str(i) for i in sel) if sel else '0,1'} - the experts the "
"GPUs do not hold are then read through the OS file cache: faster in two reports (1.3-1.6x, #364 "
f"#384), but RAM can fill up to 0 free during long prompts. Or {need:.0f} GB of RAM or more, or a "
"smaller size."]
def low_ram_together(a, model, ram, gpu, chosen) -> bool:
"""#364 #384: the low-RAM mode with several GPUs chosen. One GPU is recommended: the resident variant keeps the
experts the GPU does not hold in RAM (steady RAM use) and has no layer split. All the GPUs together read those
experts through the OS file cache instead (--mmap-experts) - 1.3-1.6x faster in those reports, but RAM can fill
up to 0 free during long prompts. An explicit --gpus (or an earlier install's cards) is kept; otherwise asked,
one GPU by default (--yes: one GPU, as before). True: all of them."""
sel = [g["index"] for g in chosen]
names = " + ".join(gpu_name(g) for g in chosen)
if a.gpus:
warn(f"the low-RAM mode on {names}, as you chose (--gpus): the experts the GPUs do not hold are read through "
"the OS file cache (the resident variant has no layer split yet), and RAM can fill up to 0 free during "
"long prompts")
say(f" One GPU keeps them in RAM (steady RAM use, recommended): --gpu {gpu['index']}")
if a.low_ram == "resident":
warn("--low-ram resident has no layer split yet: the experts are read through the OS file cache instead")
return True
if a.low_ram != "resident" and not a.yes:
say()
for line in low_ram_one_gpu_why(model, ram, a.low_ram, sel):
say(" " + line)
say(f" 1) {gpu_name(gpu)} only: the experts it does not hold kept in RAM where they fit (recommended: "
"steady RAM use)")
say(f" 2) {names} together: the experts the GPUs do not hold read through the OS file cache - faster")
say(" in two reports (1.3-1.6x, #364 #384), but RAM can fill up to 0 free during long prompts")
if ask("Low-RAM mode: which GPUs?", ["1", "2"], "1", a.yes) == "2":
ok(f"the low-RAM mode on {names}: the experts read through the OS file cache, as you chose")
return True
warn("the low-RAM mode: using " + gpu_name(gpu) + " only")
return False
warn("the low-RAM mode: using " + gpu_name(gpu) + " only (recommended)")
for line in low_ram_one_gpu_why(model, ram, a.low_ram, sel):
say(" " + line)
return False
def unsloth_together(a, model, ram, gpu, chosen) -> bool:
"""#498: UD-Q4_K_XL with several GPUs chosen. Its RAM budget (--resident-budget-gib) has no layer split, so a
split runs without it: all its experts loaded into RAM from the GGUFs at start, as with the 2-3-bit models - only
where the RAM holds the GGUF files and 24 GB more (unsloth_split_need_gb; 165 GiB, 2x RTX 3090: 31 -> 64-78
tok/s). An explicit --gpus is honoured there; otherwise asked, one GPU by default (--yes: one GPU, as before); an
explicit --resident-budget-gib keeps one GPU. True: all of them."""
names = " + ".join(gpu_name(g) for g in chosen)
need = unsloth_split_need_gb(model)
if ram < need:
warn(f"{model} runs on one GPU here: on several it has no RAM budget and needs ~{need:.0f} GB of RAM (its GGUF "
f"files and {UNSLOTH_RAM_LEFT_GB} GB more), this PC has {ram:.0f} - using {gpu_name(gpu)} only")
return False
if a.resident_budget_gib is not None:
warn(f"--resident-budget-gib has no layer split: {model} runs on one GPU with it - using {gpu_name(gpu)} only "
f"(leave the budget out to use {names} together)")
return False
note = (f"no RAM budget - all of its experts (~{MODELS[model]['arena_gb']:.0f} GB) are loaded into RAM from the "
f"model files at start, and the files pass through the OS file cache (needs ~{need:.0f} GB of RAM, this "
f"PC has {ram:.0f})")
if a.gpus:
ok(f"{model} on {names}, as you chose (--gpus): {note}")
return True
if not a.yes:
say()
say(f" {model} can run on one GPU with a RAM budget of its experts, or on {names} together without one:")
say(f" 1) {gpu_name(gpu)} only: the most-used experts kept in RAM, the rest read from the SSD (recommended: "
"the tested setup)")
say(f" 2) {names} together: {note};")
say(" about twice as fast in #498 (2x RTX 3090: 31 -> 64-78 tokens/s)")
if ask(f"{model}: which GPUs?", ["1", "2"], "1", a.yes) == "2":
ok(f"{model} on {names}: {note}")
return True
warn(f"{model} runs on one GPU: using {gpu_name(gpu)} only (--gpus " + ",".join(str(g["index"]) for g in chosen) +
f" shares it across {names} without the RAM budget: this PC's RAM holds it)")
return False
def confirm_risk(msg, explicit, yes, stop, hint=None, question=" Go on anyway?", default="n") -> None:
"""The owner's rule: setup recommends, it never forces. A choice setup expects to fail or run badly is said
plainly (msg), then asked (default n; `default` keeps an older question's own default), or with --yes taken as
consent when it was asked for explicitly (a flag such as --model or --gpus): --yes alone keeps the stop
(stop, hint). Returns when it goes on; the caller says what it does."""
warn(msg)
if yes and explicit:
return
if ask(question, ["y", "n"], default, yes) != "y":
fail(stop, hint)
def confirm_paging(model, ram, choice, yes, explicit_model=False):
"""The model's experts do not fit this PC's RAM and the low-RAM mode is off. #125: a warning and a question, not
a stop - the user may accept paging. Asked "no" by default, so an unattended --yes install stops here, unless
the low-RAM mode was turned off explicitly (--low-ram off, #250) or the size was (--model): that is the choice
already made."""
need_gb, arena = MODELS[model]["ram_gb"], MODELS[model]["arena_gb"]
off = choice == "off"
confirm_risk(f"{model} needs about {need_gb} GB of RAM and this PC has {ram:.0f} GB: its experts alone are "
f"{arena:.0f} GB and must stay in RAM, so Windows/Linux will page part of them from disk. Expect it "
"to be much slower, and it may not start at all.\n A smaller size (Q2_0 or IQ2_XS) fits; more "
"RAM fixes it.", off or explicit_model, yes,
f"{model} needs about {need_gb} GB of RAM; this PC has {ram:.0f} GB",
"choose Q2_0 or IQ2_XS, or add RAM" + ("" if off else f"; or --model {model} --yes (or --low-ram off "
"--yes) to install it anyway"),
" Install it anyway?", "y" if off else "n")
warn(f"installing {model} with {ram:.0f} GB of RAM, as you chose" + (" (--low-ram off)" if off else
" (--model)" if explicit_model else ""))
def ctx_ram_need(model, ctx, low_ram=False):
"""#406: the RAM (GB) setup estimates for a long context with IQ3_XXS / IQ3_S: their experts + the context's
8-bit KV cache + 24 GB of room for everything else (the 0.1.29 arithmetic, counted). None where the context does
not count against RAM by this rule: the other sizes, and the low-RAM mode (its KV cache stays in VRAM)."""
if model not in ("IQ3_XXS", "IQ3_S") or low_ram:
return None
return MODELS[model]["arena_gb"] + ctx * 13 * 1056 / 1e9 + 24
def ram_ctx(model, ram, low_ram=False) -> int:
"""#406: the longest context the RAM rule recommends: 128K, or longer where the estimate fits this PC's RAM. It
is part of the recommended default (the smaller of it and the GPU's rule); a longer choice is kept, with a note."""
return max(c for c in CONTEXTS if c <= 131072 or (ctx_ram_need(model, c, low_ram) or 0) <= ram)
def settings_path() -> Path:
if WIN:
return Path(os.environ.get("APPDATA") or Path.home() / "AppData" / "Roaming") / "Strata" / "settings.json"
return Path(os.environ.get("XDG_CONFIG_HOME") or Path.home() / ".config") / "strata" / "settings.json"
def load_settings() -> dict:
try:
return json.loads(settings_path().read_text(encoding="utf-8"))
except (OSError, ValueError):
return {}
def save_settings(s: dict) -> None:
try:
settings_path().parent.mkdir(parents=True, exist_ok=True)
settings_path().write_text(json.dumps(s, indent=1), encoding="utf-8")
except OSError as e:
warn(f"could not save {settings_path()} ({e})")
def has_data(folder: Path) -> bool:
for d in DATA_ITEMS:
try:
if (folder / d).is_dir() and any((folder / d).iterdir()):
return True
except OSError:
pass
return False
def other_installs(settings: dict) -> list:
"""Strata folders besides this one that may hold model files: the ones this user ran before, and Strata* folders
next to this one (a zip unpacked again lands in e.g. `Strata-main (1)\\Strata-main`)."""
cands = [Path(p) for p in settings.get("installs", [])]
for base in dict.fromkeys((ROOT.parent, ROOT.parent.parent)):
try:
for d in base.iterdir():
if d.is_dir() and d.name.lower().startswith("strata"):
cands.append(d)
cands += [c for c in d.iterdir() if c.is_dir() and c.name.lower().startswith("strata")]
except OSError:
pass
found = []
for d in cands:
try:
d = d.resolve()
if d != ROOT and d not in found and (d / "setup.py").is_file():
found.append(d)
except OSError:
pass
return found
def same_drive(a: Path, b: Path) -> bool:
try:
return os.stat(a).st_dev == os.stat(b).st_dev
except OSError:
return False
def move_into(src: Path, dst: Path) -> None:
"""A rename into the data folder (same drive: instant); a folder merges into one already there, keeping what the
destination has. Whatever cannot be moved (a file in use) stays where it is."""
if not dst.exists():
try:
dst.parent.mkdir(parents=True, exist_ok=True)
os.replace(src, dst)
return
except OSError:
if not src.is_dir():
return
dst.mkdir(parents=True, exist_ok=True)
if src.is_dir() and dst.is_dir():
for c in list(src.iterdir()):
move_into(c, dst / c.name)
try:
src.rmdir()
except OSError:
pass
def repoint_config(cfg_file: Path, old: Path, new: Path) -> None:
"""A config whose model files moved from `old` to `new` points at them there (each path only if its file is
now there and no longer at the old place)."""
try:
cfg = json.loads(cfg_file.read_text(encoding="utf-8-sig"))
except (OSError, ValueError):
return
def fix(v):
if isinstance(v, list):
return [fix(x) for x in v]
if isinstance(v, dict):
return {k: fix(x) for k, x in v.items()}
if isinstance(v, str):
for d in DATA_ITEMS:
o = str(old / d)
nv, no = os.path.normcase(v), os.path.normcase(o) # Windows: C:\ and c:\ are the same place
if nv == no or nv.startswith(no + os.sep):
n = str(new / d) + v[len(o):]
if Path(n).exists() and not Path(v).exists():
return n
return v
new_cfg = fix(cfg)
if new_cfg != cfg:
write_config(cfg_file, new_cfg)
def data_folder(requested: str | None) -> tuple:
"""(the data folder, folders on other drives that still hold model files). Moves the model files of this folder
and of earlier Strata folders on the same drive into the data folder, and points their configs there."""
settings = load_settings()
dest = Path(requested).expanduser().resolve() if requested else \
Path(settings["data_dir"]) if settings.get("data_dir") else ROOT.parent / "Strata-data"
try:
dest.mkdir(parents=True, exist_ok=True)
except OSError as e: # e.g. no write access next to the Strata folder
warn(f"cannot use {dest} for the model files ({e}): keeping them in {ROOT}")
dest = ROOT
elsewhere = []
# #198: the data folder remembered before (a --data-dir to a new place) is a source too, and so is a Strata-data
# folder nested in any of them (an install that kept its models one level down)
sources = [ROOT, *other_installs(settings)]
if settings.get("data_dir") and Path(settings["data_dir"]) != dest:
sources.append(Path(settings["data_dir"]))
sources += [f / "Strata-data" for f in list(sources) if (f / "Strata-data") != dest]
seen = set()
for folder in sources:
key = os.path.normcase(str(folder))
if key in seen:
continue
seen.add(key)
if folder == dest or not has_data(folder):
continue
if not same_drive(folder, dest):
elsewhere.append(folder) # another drive: used where it is (no 70 GB copy)
continue
# the downloads merge file by file (the same file wherever it came from); a prepared pack or MTP layer moves
# whole or not at all, so two copies are never mixed
if (folder / "models").is_dir():
move_into(folder / "models", dest / "models")
for item in [*((folder / "packs").glob("*") if (folder / "packs").is_dir() else []), folder / "mtp"]:
rel = item.relative_to(folder)
if item.exists() and not (dest / rel).exists():
move_into(item, dest / rel)
for d in ("packs",):
try:
(folder / d).rmdir() # empty now
except OSError:
pass
for c in folder.glob("strata-*.json"):
repoint_config(c, folder, dest)
if has_data(folder):
elsewhere.append(folder) # in use, or a copy the data folder already has
warn(f"some model files are still in {folder} (in use, or already in {dest})")
else:
ok(f"model files from {folder} moved to {dest} (a new copy of Strata finds them there)")
installs = [str(ROOT)] + [p for p in settings.get("installs", []) if p != str(ROOT) and Path(p).is_dir()]
save_settings({**settings, "data_dir": str(dest), "installs": installs[:20]})
return dest, elsewhere
def write_config(path: Path, cfg: dict):
"""A run config, written whole or not at all (#459): to a temporary file first, then moved over the old one, so
a setup stopped half-way (a closed window, a full disk) never leaves an empty strata-*.json behind."""
tmp = path.with_name(path.name + ".tmp")
tmp.write_text(json.dumps(cfg, indent=1), encoding="utf-8")
os.replace(tmp, path)
def readable_config(path: Path) -> bool:
"""#459: a config that parses as a JSON object; any other gets a one-line warning naming it."""
text = None
try:
text = path.read_text(encoding="utf-8-sig")
if isinstance(json.loads(text), dict):
return True
why = "not a JSON object"
except OSError as e:
why = e.strerror or str(e)
except ValueError: # JSONDecodeError, or bytes that are not UTF-8
why = "the file is empty" if text is not None and not text.strip() else "not valid JSON"
warn(f"skipped the earlier config {path} ({why}): setting this copy up without it")
return False
def previous_config(elsewhere_first: list, settings: dict):
"""The most recently used model config of another Strata folder on this PC, for a folder that has none yet. One
that does not parse (an empty or cut-off file, #459) is skipped with a warning: the newest readable one is used,
and with none this copy is set up as a fresh install."""
cands = []
for folder in [*elsewhere_first, *other_installs(settings)]:
cands += list(folder.glob("strata-*.json"))
cands = [c for c in dict.fromkeys(cands) if c.is_file()]
return next((c for c in sorted(cands, key=lambda p: p.stat().st_mtime, reverse=True) if readable_config(c)), None)
def choices_from_config(cfg_path: Path) -> dict:
"""The setup answers a config was written with (family, size, context, KV, images, projection, network)."""
cfg = json.loads(cfg_path.read_text(encoding="utf-8-sig"))
if cellar.is_cellar(cfg): # 🍷 a 9B bottle: set up the same one again
return {"family": "cellar", "model": cfg.get("cellar_model"), "context": cfg.get("max_context"), "kv": None,
"vision": None, "esp": None, "host": cfg.get("host"), "api_key": cfg.get("api_key"),
"port": cfg.get("port"), "gpu": None, "layer_split": None, "vram_reserve_mib": None}
tag = cfg_path.stem[len("strata-"):]
family = next((f for f, d in FAMILIES.items() if d["tag"] and tag.startswith(d["tag"])), "qwen")
model = (tag[len(FAMILIES[family]["tag"]):] if tag.startswith(FAMILIES[family]["tag"]) else tag).upper()
if model not in MODELS: # (sizes have no dash except UD-Q4_K_XL: the old rule)
model = tag.split("-")[-1].upper()
a = cfg.get("args", [])
val = lambda k: a[a.index(k) + 1] if k in a and a.index(k) + 1 < len(a) else None # noqa: E731
vis = cfg.get("vision")
esp = val("--control-vector-scaled")
esp_path = esp.rsplit(":", 1)[0] if esp else None
return {"family": family, "model": model if model in MODELS else None,
"context": int(val("--max-context")) if val("--max-context") else None,
"kv": val("--kv") if val("--kv") in ("int8", "q4_0") else None,
"vision": ("gpu" if vis.get("gpu") else "cpu") if isinstance(vis, dict) else "none",
"esp": ("on" if Path(esp_path).name == ESP_VECTOR.name else esp_path) if esp_path else "off",
"host": cfg.get("host"), "api_key": cfg.get("api_key"), "port": cfg.get("port"), "gpu": cfg.get("gpu"),
"layer_split": cfg.get("layer_split"),
# #493: --vram-reserve-mib given at setup (images write the default 700 themselves)
"vram_reserve_mib": int(val("--vram-reserve-mib")) if (val("--vram-reserve-mib") or "").isdigit() and (
vis is None or int(val("--vram-reserve-mib")) != VISION["gpu"]["reserve_mib"]) else None}
def find_in(roots: list, rel: str):
"""The first of roots/rel that exists."""
for r in roots:
if (r / rel).exists():
return r / rel
return None
# ------------------------------------------------------------------------------------------------ start
def model_config(path: Path) -> bool:
"""#549: a model's run config (a JSON object with "exe" and "args"). Any other strata-*.json in the folder (a
file of the user's own, a cut-off one) is skipped with a warning naming it instead of stopping setup."""
try:
cfg = json.loads(path.read_text(encoding="utf-8-sig"))
if isinstance(cfg, dict) and cfg.get("exe") and isinstance(cfg.get("args"), list):
return True
why = 'no "exe" or "args"'
except OSError as e:
why = e.strerror or str(e)
except ValueError:
why = "not valid JSON"
warn(f"skipped {path.name} ({why}): it is not a Strata model config")
return False
def installed_configs():
return [p for p in sorted(ROOT.glob("strata-*.json"), key=lambda p: p.stat().st_mtime, reverse=True)
if model_config(p)]
def source_version() -> str:
"""The engine version the source tree builds (CMakeLists.txt's project version)."""
m = re.search(r"project\(strata VERSION ([\d.]+)", (ROOT / "CMakeLists.txt").read_text(encoding="utf-8"))
return m.group(1) if m else "0"
def engine_version(exe: Path) -> tuple:
"""The version in the engine folder's BUILD.json, or else the one compiled into the binary. A locally compiled
engine is not necessarily the source's version: when compiling a `git pull` fails, the previous engine is kept
(issue #49)."""
try:
meta = json.loads((Path(exe).parent / "BUILD.json").read_text())
except (OSError, ValueError):
meta = {}
v = str(meta.get("version") or "")
if not v: # the version compiled into the binary: 0.1.13 and newer
try: # carry it, so a binary without it is older
m = re.search(rb"engine=(\d+\.\d+\.\d+)\n", Path(exe).read_bytes())
v = m.group(1).decode() if m else "0.1.12"
except OSError:
v = "0"
return tuple(int(x) for x in v.split(".")[:3] if x.isdigit())
def is_wsl() -> bool:
return sys.platform.startswith("linux") and "microsoft" in platform.uname().release.lower()
def hardware_key(cfg: dict) -> str:
"""What a calibration is valid for: this GPU, CPU and RAM, and the model with its context and images setting
(the context's KV cache and the image encoder take VRAM from the expert cache)."""
sel = cfg.get("gpu")
gl = [gpu_info(i) or {} for i in sel] if isinstance(sel, list) else [gpu_info(sel) or {}]
g = {"name": " + ".join(x.get("name", "?") for x in gl), "vram_gb": sum(x.get("vram_gb", 0) for x in gl)}
a = cfg.get("args", [])
ctx = a[a.index("--max-context") + 1] if "--max-context" in a else "?"
return "|".join([g.get("name", "?"), f"{g.get('vram_gb', 0):.0f}GB", cpu_info()[0], f"{ram_gb():.0f}GB",
cfg.get("model_name", "?"), ctx, "images" if "--vision" in a else "text"])
def calibrate_config(cfg_path: Path) -> bool:
"""Measure the engine's hardware-dependent settings on this PC (tools/calibrate.py), write them into the run
config and remember them per PC and model in the settings file, so an update or a reinstall keeps them."""
sys.path.insert(0, str(ROOT / "tools"))
import calibrate as CAL
cfg = json.loads(cfg_path.read_text(encoding="utf-8-sig"))
say()
say(" Tuning Strata for this PC: the output speed is measured with a few engine settings (the PCIe share, the")
say(" draft depth, the CPU threads). It takes about 5-10 minutes; the PC is busy meanwhile.")
try:
since = os.path.getsize(cfg["log"]) if cfg.get("log") and os.path.isfile(cfg["log"]) else 0
except OSError:
since = 0
try:
res = CAL.run(cfg, say=say)
except Exception as e: # never stops an install: the defaults stay
warn(f"the tuning did not finish ({e}): the default settings stay")
why = CAL.engine_error(cfg.get("log"), since) # #447: the engine's own reason, not only "see the log"
if why:
say(f" the engine said: {why}")
return False
cfg["args"] = CAL.apply(cfg["args"], res["settings"])
write_config(cfg_path, cfg)
st = load_settings()
st.setdefault("calibration", {})[hardware_key(cfg)] = {"settings": res["settings"], "tok_s": res["report"].get("tok_s"),
"date": time.strftime("%Y-%m-%d")}
save_settings(st)
if res["settings"]:
ok("tuned for this PC: " + ", ".join(f"{k} {v}" for k, v in res["settings"].items())
+ (f" ({res['report']['tok_s']} tok/s)" if res["report"].get("tok_s") else ""))
else:
ok("tuned for this PC: the default settings are already the fastest here"
+ (f" ({res['report']['tok_s']} tok/s)" if res["report"].get("tok_s") else ""))
return True
def saved_calibration(cfg: dict) -> dict | None:
"""The settings an earlier calibration found for this PC and model, if any."""
return (load_settings().get("calibration") or {}).get(hardware_key(cfg))
def upgrade_config(cfg_path: Path, cfg: dict) -> dict:
"""Configs written before v0.1.13 read prompts in fixed 2048-token chunks; the engine now picks the chunk
itself (`--prefill auto`: up to 8192, as the free VRAM allows - about 2x faster on long prompts). Under WSL,
KV streaming is dropped: its RAM copy must be pinned, and the driver pins only about 1 GB there."""
a = cfg.get("args", [])
changed = False
ver = engine_version(cfg["exe"]) if "--prefill" in a else (0, 0, 0)
if "--prefill" in a and a[a.index("--prefill") + 1] == "2048" and ver >= (0, 1, 13):
a[a.index("--prefill") + 1] = "auto"
changed = True
ok("prompt reading: the engine now picks its chunk size (--prefill auto)")
elif "--prefill" in a and a[a.index("--prefill") + 1] == "auto" and (0, 0, 0) < ver < (0, 1, 13):
a[a.index("--prefill") + 1] = "2048" # an older engine kept after a failed update (issue #49)
changed = True
warn(f"the installed engine is {'.'.join(map(str, ver))}: prompts are read in 2048-token chunks until it is updated")
if is_wsl() and "--kv-resident" in a:
i = a.index("--kv-resident")
del a[i:i + 2]
changed = True
ok("WSL: KV streaming off (the driver pins only about 1 GB of RAM); the KV cache stays in VRAM")
if changed:
write_config(cfg_path, cfg)
return cfg
def update_install(have: list, a) -> int:
"""#475: `setup.py --update` (UPDATE.bat / update.sh, after their git pull): what a plain START-HERE.bat does to
an install before it starts the model, without starting it - the Python packages, the ready-made engine when this
setup needs a newer one (MIN_ENGINE; a compiled engine when its source changed), each installed model's config
upgrades and its draft subset. No question is asked and the model files are not touched; a model still running
keeps its engine (update_installed_engine says to close it and run this again)."""
have = [p for p in have if model_config(p)] # #549: a strata-*.json that is no model config is skipped
if not have:
say(" No model is installed in this Strata folder yet: run START-HERE.bat (Linux: ./setup.sh) to set it up -")
say(" it finds an earlier install's model files next to it and reuses them.")
return 0
pip_install(requirement_lines() if REQUIREMENTS.exists() else PY_PACKAGES,
"numpy, jinja2, regex, pyyaml, tqdm, requests, cmake, ninja, pillow, psutil")
if not a.build:
update_installed_engine(a.prebuilt)
for cfg_path in have:
if cellar.is_cellar(json.loads(cfg_path.read_text(encoding="utf-8-sig"))):
ok(f"{cfg_path.stem}: up to date (llama.cpp {cellar.LLAMA_BUILD})")
continue
cfg = upgrade_config(cfg_path, json.loads(cfg_path.read_text(encoding="utf-8-sig")))
if "--mtp" in cfg["args"][:-1]:
refresh_draft_vocab(Path(cfg["args"][cfg["args"].index("--mtp") + 1]), cfg.get("draft_vocab", "cjk"))
if cfg.get("backend") == "hip" and WIN:
hip_runtime_beside_exe(Path(cfg["exe"]).parent) # #468 #461
ok(f"{cfg.get('model_name', cfg_path.stem)}: up to date")
strata_cfgs = [p for p in have if not cellar.is_cellar(json.loads(p.read_text(encoding="utf-8-sig")))]
ver = engine_version(Path(json.loads(strata_cfgs[0].read_text(encoding="utf-8-sig"))["exe"])) if strata_cfgs \
else ()
say()
ok("Strata is updated" + (f" (engine {'.'.join(map(str, ver))})" if any(ver) else "") +
". Start the model with " + ("START-HERE.bat" if WIN else "./setup.sh") + " when you want it.")
return 0
def settings_summary(cfg: dict, port=None) -> str:
"""#564: the settings a start uses, in one line: the config's engine options (the model's file paths left out)
and the server's own fields, so a change made by hand to strata-<model>.json can be checked without the log."""
a, out, i = [str(x) for x in cfg.get("args") or []], [], 0
while i < len(a):
flag = a[i]
val = a[i + 1] if i + 1 < len(a) and not a[i + 1].startswith("--") else None
i += 1 if val is None else 2
if not flag.startswith("--"):
continue # a positional: the model file
if val is not None and ("/" in val or "\\" in val or val.lower().endswith((".gguf", ".bin"))):
continue # a path: --native, --mtp, --profile ...
out.append(flag if val is None else f"{flag} {val}")
srv = [f"{cfg.get('host', '127.0.0.1')}:{port or cfg.get('port', 8080)}"]
if cfg.get("api_key"):
srv.append("api key set")
for k in ("gpu", "layer_split", "draft_vocab", "fit_max_tokens", "reasoning_budget_tokens", "anthropic_thinking"):
if cfg.get(k) is not None:
v = cfg[k]
srv.append(f"{k} {','.join(map(str, v)) if isinstance(v, list) else str(v).lower() if isinstance(v, bool) else v}")
return " ".join(out) + ("; " if out else "") + "server " + ", ".join(srv)
def start(cfg_path: Path, port: int | None, gpu: int | list | None = None, open_browser=True, yes=False,
layer_split=None, keep=None) -> int:
"""keep: settings given on this start that the model keeps from now on (--host, --api-key, --draft-vocab,
--vram-reserve-mib)."""
if cellar.is_cellar(json.loads(cfg_path.read_text(encoding="utf-8-sig"))): # 🍷 a 9B bottle: llama.cpp
return cellar.start(sys.modules[__name__], cfg_path, port, open_browser, keep)
cfg = upgrade_config(cfg_path, json.loads(cfg_path.read_text(encoding="utf-8-sig")))
missing = [p for p in [cfg["exe"], *[a for a in cfg["args"] if a.endswith(".gguf")]] if not Path(p).exists()]
if missing:
fail(f"{cfg_path.name} refers to missing files: {missing[0]}", "run it again with --setup to repair")
keep = {k: v for k, v in (keep or {}).items() if v is not None}
reserve = keep.pop("vram_reserve_mib", None) # #493: an engine argument, kept in the config's args
if reserve is not None:
args = cfg["args"]
if "--vram-reserve-mib" in args[:-1]:
args[args.index("--vram-reserve-mib") + 1] = str(reserve)
else:
args += ["--vram-reserve-mib", str(reserve)]
write_config(cfg_path, cfg)
ok(f"saved for this model: {reserve} MiB of VRAM kept free for other programs (--vram-reserve-mib)")
if keep and any(cfg.get(k) != v for k, v in keep.items()): # #179: a --host/--api-key on a start was ignored
cfg.update(keep)
write_config(cfg_path, cfg)
ok("saved for this model: " + ", ".join("api key" if k == "api_key" else f"{k.replace('_', ' ')} {v}"
for k, v in keep.items()))
cfg_path.touch() # the most recently used model
if "--mtp" in cfg["args"][:-1]:
refresh_draft_vocab(Path(cfg["args"][cfg["args"].index("--mtp") + 1]), cfg.get("draft_vocab", "cjk"))
cmd = [sys.executable, str(ROOT / "serve" / "server.py"), "--engine", "strata", "--config", str(cfg_path),
"--port", str(port or cfg.get("port", 8080))]
if cfg.get("backend") == "hip": # AMD, numbered as HIP numbers them (setup's KFD order)
if WIN:
hip_runtime_beside_exe(Path(cfg["exe"]).parent) # #468 #461: also fixes a 0.1.34 install
amd = amd_gpus()
if isinstance(gpu, list): # --gpus: saved, this model runs on these cards from now on
cards = amd_parse_gpus(",".join(str(i) for i in gpu), amd)
built = (engine_archs_hip() or [])
miss = [x for x in cards if built and x["arch"] not in built]
if miss:
fail("the installed engine has no code for " + ", ".join(f"{x['name']} ({x['arch']})" for x in miss),
"set it up for these cards: ./setup.sh --setup --backend hip --gpus " + ",".join(map(str, gpu)))
cfg["gpu"], cfg["gpus_asked"] = gpu, True
cfg["layer_split"] = layer_split or cfg.get("layer_split") or "auto"
split_budget(cfg) # #498: before it is saved (it stops when the RAM is short)
write_config(cfg_path, cfg)
gpu = None
elif gpu is not None:
cmd += ["--gpu", str(gpu)]
found = []
use = gpu if gpu is not None else cfg.get("gpu")
if isinstance(use, list):
byid = {x["index"]: x for x in amd}
ok("GPUs: " + " + ".join(gpu_name(byid[i]) if i in byid else f"GPU {i} (not found)" for i in use)
+ f" together, AMD (layers split {cfg.get('layer_split') or 'auto'})")
else:
g = next((x for x in amd if x["index"] == (use if use is not None else x["index"])
and amd_problem(x) is None), None)
if g is not None:
ok(f"GPU: {g['name']} ({g['vram_gb']:.0f} GB, AMD)")
else:
found = gpus()
if cfg.get("backend") == "hip":
pass
elif isinstance(gpu, list): # --gpus: saved, this model runs on these cards from now on
check_gpus(gpu, found, yes=yes, named=True)
cfg["gpu"], cfg["gpus_asked"] = gpu, True
cfg["layer_split"] = layer_split or cfg.get("layer_split") or "auto"
split_budget(cfg) # #498: before it is saved (it stops when the RAM is short)
write_config(cfg_path, cfg)
gpu = None
elif gpu is not None: # --gpu N: this start only, on that card
check_gpus([gpu], found)
cmd += ["--gpu", str(gpu)]
else:
cfg = offer_together(cfg_path, cfg, yes)
use = gpu if gpu is not None else cfg.get("gpu")
# #364 #384: a resident low-RAM config on several GPUs; #498: a UD-Q4_K_XL config with its RAM budget (by hand)
if isinstance(use, list) and (split_mmap(cfg) | split_budget(cfg)):
write_config(cfg_path, cfg)
if cfg.get("backend") == "hip":
pass
elif isinstance(use, list):
check_gpus(use, found, "(chosen for this model) ", yes=True, named=True)
byid = {g["index"]: g for g in found}
cfg = ensure_engine_for([byid[i] for i in use], cfg_path, cfg, yes)
ok("GPUs: " + " + ".join(gpu_name(byid[i]) for i in use) + f" together (layers split {cfg.get('layer_split') or 'auto'})")
elif found:
g = next((x for x in found if x["index"] == use), None) if use is not None else max(
found, key=lambda x: (round(x["vram_gb"]), -x["index"]))
if g is not None:
cfg = ensure_engine_for([g], cfg_path, cfg, yes)
ok("GPU: " + gpu_name(g))
if open_browser:
cmd.append("--open")
gb = 0.0
if "--native" in cfg["args"]:
try:
gb = Path(cfg["args"][cfg["args"].index("--native") + 1]).stat().st_size / 1e9
except (OSError, IndexError):
pass
say()
say(" " + "-" * 100)
size = f'about {gb:.0f} GB' if gb >= 1 else '34-55 GB'
a_ = cfg["args"]
if "--mmap-experts" in a_ and "--resident-budget-gib" not in a_ and "--resident-experts" not in a_:
# #505: the mapped low-RAM mode loads nothing into RAM up front (the server's narrator says the same)
say(f" Starting {cfg.get('model_name', 'the model')}: it maps {size} of experts from the model files (the OS "
"file cache reads them).")
else:
say(f" Starting {cfg.get('model_name', 'the model')}: it loads {size} into RAM and locks part of it for the "
"GPU.")
say(" While it does, YOUR PC CAN BE SLOW OR STOP RESPONDING FOR 1-3 MINUTES (longer the first time after a")
say(" restart). That is normal: please wait and don't close this window - the browser opens when it is ready.")
say(" Later, closing this window stops the model.")
say(" " + "-" * 100)
for n, line in enumerate(textwrap.wrap(f"Settings ({cfg_path.name}): {settings_summary(cfg, port)}", 100,
break_on_hyphens=False)): # #564: what this start uses
say((" " if n == 0 else " ") + line)
if not WIN and os.environ.get("STRATA_EXECV"):
# Replace this process instead of spawning a child. The Docker image sets STRATA_EXECV=1,
# so there the server is PID 1 and docker stop's SIGTERM reaches the process that can
# answer the engine with QUIT. Normal Linux starts keep spawning the server as a child.
os.execv(cmd[0], cmd)
return subprocess.call(cmd)
# the draft subsets setup copied before (sha256): replaced by the current one, a subset made by hand is kept
OLD_DRAFT_VOCABS = {"369151522226a5edaa5f12cfd1e2ae7db8f4fbdbd222f3dcf327dced9597fb25"} # to 0.1.26: 27 Han tokens
DRAFT_VOCABS = {"cjk": "draft_vocab.bin", "en": "draft_vocab_en.bin", "cyrillic": "draft_vocab_cyrillic.bin"}
def saved_draft_vocab(cfg_path: Path) -> str | None:
"""The draft subset a model's config chose earlier (--draft-vocab), or None: a setup run again without the flag
rewrites the config, and would otherwise put the default subset back."""
try:
v = json.loads(cfg_path.read_text(encoding="utf-8-sig")).get("draft_vocab")
except (OSError, ValueError, AttributeError):
return None
return v if v in DRAFT_VOCABS else None
DRAFT_VOCAB_MIB = {"cjk": 348, "cyrillic": 193, "en": 133} # the draft head's VRAM per subset (IQ3_S: the largest)
SMALL_DRAFT_VRAM_GB = 14 # #474: below this the default subset's head can be what does not fit
def draft_vocab_note(vram_gb: float, chosen: str | None) -> list[str]:
"""#474: on a card under 14 GB, the default draft subset (cjk, ~348 MiB of VRAM) can be what does not fit at the
start ("the draft head does not fit"), and the engine's expert cache gets what a smaller one leaves. Setup
RECOMMENDS a smaller one here and changes nothing (the owner's rule, #403 #406): a subset chosen with
--draft-vocab, or kept from an earlier install, gets no note. [] for every other case."""
if chosen or not 0 < vram_gb < SMALL_DRAFT_VRAM_GB:
return []
start = "START-HERE.bat" if WIN else "./setup.sh"
return [f"Tip for a {vram_gb:.0f} GB card: the draft layer's default token subset (with Chinese, Japanese and "
f"Korean) needs up to ~{DRAFT_VOCAB_MIB['cjk']} MiB of VRAM.",
f" For English and code answers, {start} --draft-vocab en needs up to ~{DRAFT_VOCAB_MIB['en']} MiB "
f"(cyrillic: ~{DRAFT_VOCAB_MIB['cyrillic']}) and leaves the rest to the expert cache - and it is the",
" fix when the start stops with \"the draft head does not fit\". The model keeps the choice."]
POCKET_VRAM_GB = 10 # 🍷 below this, setup recommends Winery Pocket for the winery family
SMALL_CARD_GB = 7.5 # #496: a card under 8 GB gets a tip (an 8 GB card lists 7.99)
def small_card_note(ctx: int, draft_vocab: str | None) -> list[str]:
"""#496: what frees VRAM on a card under 8 GB when the start stops with "no VRAM is left for the expert cache"
(the engine already lowers its own reserve on such a card) - a recommendation, setup changes none of it. (The
draft layer stays: the server needs it.)"""
start = "START-HERE.bat --setup" if WIN else "./setup.sh"
tips = []
if ctx > 8192:
tips.append("an 8K context (a smaller KV cache)")
if draft_vocab != "en":
tips.append(f"--draft-vocab en (a draft head of ~{DRAFT_VOCAB_MIB['en']} MiB instead of "
f"~{DRAFT_VOCAB_MIB[draft_vocab or 'cjk']})")
lines = ["If the start stops with \"no VRAM is left for the expert cache\" (the engine's log says how much is "
"short):"]
if tips:
lines.append(f" run {start} again with " + " and ".join(tips) + ", or close other programs that use the GPU.")
else:
lines.append(" close other programs that use the GPU.")
return lines
DESKTOP_RESERVE_MIB = 3072 # #560 #516: what kept a KDE/Wayland desktop alive beside a full expert cache
def linux_desktop(env=None) -> bool:
"""A graphical session on Linux (Wayland or X)."""
env = os.environ if env is None else env
return sys.platform.startswith("linux") and bool(env.get("WAYLAND_DISPLAY") or env.get("DISPLAY"))
def desktop_reserve_note() -> list[str]:
"""#560 #516: an AMD card that also drives a Linux desktop - with the default 700 MiB reserve the expert cache
fills it, and when the desktop needs more VRAM amdgpu moves the cache to system RAM, where the OOM killer then ends
the compositor. A recommendation, setup changes nothing."""
return [f"If this AMD card also drives your desktop and the desktop or apps crash once the model is loaded, keep "
f"more VRAM free: ./setup.sh --vram-reserve-mib {DESKTOP_RESERVE_MIB}",
" (remembered for this model; the expert cache gets ~2.3 GB less, a few % of speed)"]
def mtp_corrupt(mtp: Path, env=None) -> bool:
"""#327: True when the MTP tensors an install fetched are not the pinned checkpoint's (tools/mtp_fetch.py verify,
which hashes only files that changed since they last checked out). A mirror that ignored range requests left the
shards' starts there instead, and the draft layer built from them accepted nothing - with no error anywhere."""
if not (mtp / "tensors").is_dir():
return False
r = subprocess.run([sys.executable, str(ROOT / "tools" / "mtp_fetch.py"), "verify", "--out", str(mtp)], env=env,
stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True)
return r.returncode == 3
def refresh_draft_vocab(rt: Path, choice: str = "cjk") -> None:
"""The draft layer's token subset in the MTP folder: `cjk` (data/draft_vocab.bin, since 0.1.27, #137), `en`
(data/draft_vocab_en.bin, the English/code subset before it: ~110 MiB less VRAM, English answers 1-2% faster) or
`cyrillic` (data/draft_vocab_cyrillic.bin: English/code and the whole Cyrillic script, for Ukrainian, Russian,
Bulgarian, Serbian... answers).
Copied when missing or when a shipped subset other than the chosen one is there; a subset made by hand is kept."""
new, dst = ROOT / "data" / DRAFT_VOCABS.get(choice, "draft_vocab.bin"), rt / "draft_vocab.bin"
if not new.exists() or not rt.is_dir():
return
if dst.exists():
old = hashlib.sha256(dst.read_bytes()).hexdigest()
shipped = OLD_DRAFT_VOCABS | {hashlib.sha256((ROOT / "data" / f).read_bytes()).hexdigest()
for f in DRAFT_VOCABS.values() if (ROOT / "data" / f).exists()}
if old not in shipped or old == hashlib.sha256(new.read_bytes()).hexdigest():
return
ok("draft layer: the token subset " + {"cjk": "with Chinese, Japanese and Korean",
"cyrillic": "with the Cyrillic script"}.get(choice,
"for English and code (less VRAM)"))
shutil.copyfile(new, dst)
def ensure_engine_for(cards, cfg_path: Path, cfg: dict, yes: bool) -> dict:
"""The installed engine must have code for every card the model starts on: a card added later (--gpus with an
older or newer generation, #128) or a new GPU in the PC otherwise stops the start with 'no kernel image'. Such a
card gets the engine compiled for all of them, before the start."""
missing = [g for g in cards if not engine_runs_on(g)]
if not missing:
return cfg
info = ROOT / "engine" / "BUILD.json"
meta = json.loads(info.read_text())
say()
say(" The installed engine has no code for " + ", ".join(f"{g['name']} (sm_{g['arch']})" for g in missing) +
": it is compiled for " + ("these cards" if len(cards) > 1 else "it") + " now.")
main = gpu_info(cards[0]["index"])
archs = sorted({int(x) for x in meta.get("archs", [])} | {int(g["arch"]) for g in cards})
vision = meta.get("vision") or ("gpu" if (ROOT / "engine" / VEXE).exists() else "none")
build_engine({**main, "archs": archs}, vision, yes, get_llama_cpp())
dirs = json.loads(info.read_text()).get("cuda_dirs") or []
cfg["lib_dirs"] = dirs + [d for d in cfg.get("lib_dirs") or [] if d not in dirs]
write_config(cfg_path, cfg)
return cfg
def write_run_script(model, cfg_path, port):
serve = [sys.executable, str(ROOT / "serve" / "server.py"), "--engine", "strata", "--config", str(cfg_path),
"--port", str(port), "--open"]
if WIN:
script = ROOT / f"run-{model.lower()}.bat"
script.write_text("@echo off\r\ntitle Strata " + model + "\r\ncd /d \"" + str(ROOT) + "\"\r\n" +
" ".join(f'"{x}"' for x in serve) + "\r\nif errorlevel 1 pause\r\n", encoding="utf-8")
else:
script = ROOT / f"run-{model.lower()}.sh"
script.write_text("#!/bin/sh\ncd \"" + str(ROOT) + "\"\nexec " + " ".join(f'"{x}"' for x in serve) + "\n",
encoding="utf-8")
script.chmod(0o755)
return script
# ------------------------------------------------------------------------------------------------ the rope config
def derived_factor(ctx: int, trained: int = 262144) -> float:
"""The automatic extension factor: the FINAL context over the trained one, at least 1.
Factor 1 removes the automatic expansion - the trained angles stand as they are - but it is not a
switch for rope as a whole: an explicitly chosen method's settings keep their defined behavior.
"""
return max(1.0, float(ctx) / float(trained))
def resolve_rope(ctx: int, scaling, scale, trained: int = 262144):
"""The rope config for the context ACTUALLY SERVED: (scaling, scale); scaling None = no scaling flags.
An explicit --rope-scaling/--rope-scale always wins - a user-supplied factor is kept verbatim even
when a reduction changed the context. Past the trained range an omitted method defaults to yarn -
llama.cpp's extension method: the trained angles survive on the high-frequency pairs and the
magnitude correction keeps the attention temperature - and an omitted factor is derived from the
final context (final / trained, at least 1), as is an explicitly chosen method's missing factor
inside the trained range: factor 1, the trained angles, no expansion. An explicit none is refused
past the trained range (the setup will not configure a run it knows is out of spec) rather than
silently overridden.
"""
if ctx <= trained:
if scale is not None and scaling in (None, "none"):
raise ValueError("--rope-scale needs --rope-scaling linear or yarn (the chosen context fits the "
"trained 262144, so there is nothing to scale)")
if scaling in (None, "none"):
return None, None # the stock model, by choice or by default
return scaling, scale if scale is not None else derived_factor(ctx, trained)
if scaling == "none":
raise ValueError(f"a {ctx // 1024}K context is past the model's trained 262144, and --rope-scaling none "
"keeps the stock angles there - the model has never seen those positions, so the setup "
"refuses the combination instead of quietly overriding it. Pick --rope-scaling yarn or "
"linear, or rerun with --context 262144 or lower")
return scaling or "yarn", scale if scale is not None else derived_factor(ctx, trained)
# ------------------------------------------------------------------------------------------------ main
def main() -> int:
ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
ap.add_argument("--family", choices=list(FAMILIES) + ["cellar"],
help="qwen = Qwen3.8-Flash-Next, swift = Swift 1.5, cellar = 🍷 the Winery 9B blends (llama.cpp)")
ap.add_argument("--model", choices=list(MODELS) + list(cellar.BOTTLES))
ap.add_argument("--context", type=int)
ap.add_argument("--rope-scaling", choices=["none", "linear", "yarn"],
help="the RoPE extension for a context past the model's trained 262144: linear (position "
"interpolation) or yarn - llama.cpp's types. Omitted with a scaled context, the setup "
"picks yarn; 'none' is refused for such a context")
ap.add_argument("--rope-scale", type=float,
help="the extension factor (default: the final context over the trained 262144, at least 1 - "
"1.5 for 384K, 2 for 512K, 1 inside the trained range)")
ap.add_argument("--kv", choices=["int8", "q4_0", "k8v4"],
help="KV cache precision above 8K context: int8 (default), q4_0 (half the memory, a little less "
"precise) or k8v4 (hybrid: INT8 K + 4-bit V, 816 B/cell)")
ap.add_argument("--vision", choices=["yes", "no", "none", "gpu", "cpu"],
help="let the model read images (yes = the encoder on the GPU)")
ap.add_argument("--experimental-speed-projection", metavar="on|off|GGUF",
help="EXPERIMENTAL, off by default: the control vector in data/experimental-speed-projection "
"(or another GGUF) as a projection on layers 4-44; see docs/DETAILS.md")
ap.add_argument("--port", type=int, help="the server's port (default: the one the install was set up with, 8080 for a new one)")
ap.add_argument("--gpu", help="one GPU, numbered as nvidia-smi numbers them (default: asked when several can be "
"used; with --setup it is saved, when starting it is for that start only)")
ap.add_argument("--gpus", help="several GPUs sharing one model, as nvidia-smi numbers them (AMD: as setup lists "
"them): \"0,2\", or \"all\" (every card that can); the first is the main one. "
"Saved, also when starting (see docs/MULTI_GPU.md)")
ap.add_argument("--layer-split", help="with --gpus: where each later GPU's layers start (\"18\", \"16,32\"); "
"default auto, placed from each GPU's free VRAM")
ap.add_argument("--host", help="where the server listens: 127.0.0.1 = this PC only (default), 0.0.0.0 = also other "
"devices on your network (issue #26; set --api-key too)")
ap.add_argument("--api-key", help="require this key from clients (recommended with --host 0.0.0.0)")
ap.add_argument("--data-dir", help="where the model files go (~70-120 GB): default Strata-data next to this folder, "
"remembered for every Strata folder on this PC")
ap.add_argument("--models-dir", help="where the GGUF files go (default: <data folder>/models)")
ap.add_argument("--gguf-dir", help="use GGUF files you already have (a folder with every shard: "
"<name>-00001-of-0000N.gguf ... -0000N-of-0000N.gguf)")
ap.add_argument("--yes", action="store_true", help="accept the recommended answers")
ap.add_argument("--setup", action="store_true", help="install another model or change settings")
ap.add_argument("--no-start", action="store_true", help="install only, do not start the model")
ap.add_argument("--update", action="store_true",
help="update the installed engine, Python packages and model settings as a start would, without "
"starting the model (UPDATE.bat / update.sh run it after a git pull)")
ap.add_argument("--build", action="store_true", help="compile the engine instead of using the ready-made one")
ap.add_argument("--prebuilt", default=os.environ.get("STRATA_PREBUILT_URL", PREBUILT_URL),
help="where the ready-made engine is (a URL folder or a local folder)")
ap.add_argument("--check", action="store_true", help="only check this PC and exit")
ap.add_argument("--calibrate", action="store_true",
help="tune the engine's settings for this PC (about 5-10 minutes), then start the model")
ap.add_argument("--draft-vocab", choices=list(DRAFT_VOCABS),
help="the draft layer's tokens: cjk = with Chinese, Japanese and Korean (default), en = English "
"and code only (~110 MiB less VRAM, English answers 1-2%% faster), cyrillic = English, code "
"and the Cyrillic script (Ukrainian, Russian... answers decode ~30%% faster)")
ap.add_argument("--low-ram", choices=["auto", "on", "off", "resident", "mmap"], default="auto",
help="read the model's experts from one file in its folder instead of copying them all into RAM "
"(for a PC with a big GPU and little RAM); auto: when the experts would not fit the RAM. In "
"this mode the experts the GPU does not hold are copied into RAM once when they fit (resident), "
"else read through the OS file cache (mmap); resident / mmap force one of the two")
ap.add_argument("--resident-budget-gib", type=float, metavar="N",
help="UD-Q4_K_XL: the GiB of its experts kept in RAM (default: the RAM less 24 GB, 40 on 64 GB; "
"more is kept as you choose, with a note)")
ap.add_argument("--vram-reserve-mib", type=int, metavar="N",
help="VRAM in MiB the engine leaves free for other programs (a game, another model; the engine's "
"default: 700); the expert cache takes that much less")
ap.add_argument("--kv-streaming", choices=["auto", "on", "off"], default="auto",
help="from a 64K context: keep the KV cache in RAM and only the attention's window in VRAM (more "
"experts fit on the GPU); auto: when the RAM has room for it")
ap.add_argument("--backend", choices=["cuda", "hip"],
help="cuda = NVIDIA (default), hip = AMD RX 7900 / 7800 / 7700 XT, RX 9060 XT / 9070 / AI PRO R9700 on "
"Linux or Windows (chosen by itself when the PC has no NVIDIA card Strata can use)")
ap.add_argument("--skip-build", action="store_true", help=argparse.SUPPRESS)
a = ap.parse_args()
if a.resident_budget_gib is not None and not a.resident_budget_gib > 0:
ap.error("--resident-budget-gib takes a number of GiB above 0, e.g. --resident-budget-gib 32")
if a.vram_reserve_mib is not None and a.vram_reserve_mib < 0:
ap.error("--vram-reserve-mib takes a number of MiB, 0 or more, e.g. --vram-reserve-mib 2048")
if a.gpu is not None: # --gpu 0,2 means --gpus 0,2 (a user tried it: issue report)
if "," in a.gpu:
a.gpus, a.gpu = a.gpus or a.gpu, None
elif a.gpu.strip().isdigit():
a.gpu = int(a.gpu)
else:
ap.error(f"--gpu takes a GPU number as nvidia-smi numbers them, e.g. --gpu 1 (or --gpus 0,2), not {a.gpu!r}")
say("Strata - Qwen3.8-Flash-Next on a normal PC (a GPU + system RAM + CPU)")
data, elsewhere = data_folder(a.data_dir) # the model files: in the data folder, found from any copy
roots = [data, *elsewhere]
if a.models_dir is None:
a.models_dir = str(data / "models")
# ---- 0. already installed: just start it
have = installed_configs()
if a.update: # #475: UPDATE.bat / update.sh - never starts the model
return update_install(have, a)
explicit = a.setup or a.model or a.family or a.check or a.no_start
if not have and not explicit: # a new copy of Strata (an update unzipped elsewhere): set it
prev = previous_config(elsewhere, load_settings()) # up like the last one, from the files already here
if prev is not None:
ch = choices_from_config(prev)
if ch["model"]:
say(f" Found your earlier install in {prev.parent} ({prev.stem[len('strata-'):]}): setting up this "
"copy the same way - the model files are reused, nothing big is downloaded.")
a.family, a.model, a.context = ch["family"], ch["model"], a.context or ch["context"]
a.kv = a.kv or ch["kv"]
a.vision = a.vision or ch["vision"]
a.experimental_speed_projection = a.experimental_speed_projection or ch["esp"]
a.host, a.api_key = a.host or ch["host"], a.api_key or ch["api_key"]
a.port = a.port or ch["port"]
if a.vram_reserve_mib is None: # #493: an explicit reserve set up before
a.vram_reserve_mib = ch.get("vram_reserve_mib")
if isinstance(ch.get("gpu"), list): # a layer split: set up across the same cards again
a.gpus = a.gpus or ",".join(str(g) for g in ch["gpu"])
a.layer_split = a.layer_split or ch.get("layer_split")
else:
a.gpu = a.gpu if a.gpu is not None else ch.get("gpu")
a.yes = True
global GPU_PICK
# starting an installed model: --gpus 0,2 (or all) saves those cards for it and starts on them (it used to start
# on the first one alone unless given with --setup), --gpu N runs this start on one card; neither: the saved
# choice, and asked once when the PC has cards that could share the model
run_gpu = start_gpus(a.gpus) or a.gpu
port = a.port or 8080 # a new install's port (issue #32: --port for an existing one)
if have and a.calibrate and not (a.setup or a.model or a.family or a.check):
if not a.build:
update_installed_engine(a.prebuilt)
pick_cfg = have[0]
if len(have) > 1:
say()
for i, c in enumerate(have, 1):
say(f" {i}) {json.loads(c.read_text(encoding='utf-8-sig')).get('model_name', c.stem)}")
pick_cfg = have[int(ask("Tune which one?", [str(i) for i in range(1, len(have) + 1)], "1", a.yes)) - 1]
if not calibrate_config(pick_cfg): # #447: said again where it is not lost above the start
say()
warn("this PC is NOT tuned: the tuning failed (the reason is above); the model "
+ ("keeps" if a.no_start else "starts with") + " the default settings")
return 0 if a.no_start else start(pick_cfg, a.port, run_gpu, yes=a.yes, layer_split=a.layer_split,
keep={"host": a.host, "api_key": a.api_key, "draft_vocab": a.draft_vocab,
"vram_reserve_mib": a.vram_reserve_mib})
if have and not (a.setup or a.model or a.family or a.check or a.no_start):
if not a.build:
update_installed_engine(a.prebuilt)
if len(have) == 1:
return start(have[0], a.port, run_gpu, yes=a.yes, layer_split=a.layer_split,
keep={"host": a.host, "api_key": a.api_key, "draft_vocab": a.draft_vocab,
"vram_reserve_mib": a.vram_reserve_mib})
say()
for i, c in enumerate(have, 1):
say(f" {i}) {json.loads(c.read_text(encoding='utf-8-sig')).get('model_name', c.stem)}")
say(f" {len(have) + 1}) install another model / change settings")
pick = int(ask("Which one?", [str(i) for i in range(1, len(have) + 2)], "1", a.yes))
if pick <= len(have):
return start(have[pick - 1], a.port, run_gpu, yes=a.yes, layer_split=a.layer_split,
keep={"host": a.host, "api_key": a.api_key, "draft_vocab": a.draft_vocab,
"vram_reserve_mib": a.vram_reserve_mib})
# 🍷 the Winery 9B cellar: dense models on llama.cpp, its own short setup (cellar.py)
if a.family == "cellar" or a.model in cellar.BOTTLES:
return cellar.run(a, sys.modules[__name__])
# ---- 1. the PC
step(1, "checking your PC")
found = gpus()
amd = amd_gpus()
nv_ok = any(gpu_problem(g) is None for g in found)
amd_ok = [g for g in amd if amd_problem(g) is None]
hip = a.backend == "hip" or (a.backend is None and not nv_ok and bool(amd_ok))
if a.backend is None and nv_ok and amd_ok:
# both kinds of card: asked (a first run on such a PC used to take NVIDIA without mentioning the Radeon)
say()
say(" This PC has NVIDIA and AMD cards Strata can use:")
say(" 1) NVIDIA: " + ", ".join(f"{g['name']} ({g['vram_gb']:.0f} GB)" for g in found if gpu_problem(g) is None)
+ " (recommended)")
say(" 2) AMD: " + ", ".join(f"{g['name']} ({g['vram_gb']:.0f} GB)" for g in amd_ok)
+ f" ({'the ready-made AMD engine, no images' if WIN else 'compiled here, images on the CPU'}"
" - docs/AMD_HIP.md)")
hip = ask("Which cards?", ["1", "2"], "1", a.yes or a.check) == "2"
if a.check and not hip:
say(f" (the AMD card: {'START-HERE.bat' if WIN else './setup.sh'} --backend hip)")
if hip: # AMD: compiled here; Windows: ready-made
if WIN and a.gpus:
fail("several AMD cards sharing one model (--gpus) is Linux-only for now", "use one card: --gpu N")
say(" Your AMD GPUs:" if amd else " No AMD GPU found (" + ("Windows lists no AMD display adapter)." if WIN
else "the amdgpu driver's KFD topology is empty)."))
for g in amd:
say(f" GPU {g['index']}: {g['name']}, {g['vram_gb']:.0f} GB VRAM - " + (amd_problem(g) or "can be used"))
usable = [g for g in amd if amd_problem(g) is None]
if not usable:
fail("no AMD GPU Strata can use", f"the AMD backend runs on {AMD_CARDS}")
if a.gpus: # a layer split across these cards, the first one the main
chosen = amd_parse_gpus(a.gpus, amd)
gpu = chosen[0]
elif a.gpu is not None:
gpu = next((g for g in usable if g["index"] == a.gpu), None)
if gpu is None:
fail(f"AMD GPU {a.gpu} cannot be used", "use one of: " + ", ".join(f"--gpu {g['index']}" for g in usable))
chosen = [gpu]
else:
gpu = max(usable, key=lambda x: (round(x["vram_gb"]), -x["index"]))
chosen = [gpu]
# the engine is compiled for every chosen card's architecture
gpu = {**gpu, "count": len(amd), "archs": sorted({g["arch"] for g in chosen})}
chosen = [gpu] + chosen[1:]
sel = [g["index"] for g in chosen]
multi = sel if len(sel) > 1 else []
a.gpu = gpu["index"] if len(amd) > 1 else a.gpu
if multi:
ok("GPUs: " + " + ".join(gpu_name(x) for x in chosen) + " together (the model's layers are split across them)")
ok(f"GPU: {gpu['name']}, {gpu['vram_gb']:.1f} GB VRAM, {gpu['arch']} (AMD: docs/AMD_HIP.md)")
else:
if not found:
fail("no NVIDIA GPU found (nvidia-smi did not answer)",
"install the NVIDIA driver from https://www.nvidia.com/drivers and restart the PC (or run the "
"🍷 Winery 9B cellar, any GPU or the CPU: --family cellar)"
+ (f"; AMD ({', '.join(AMD_ARCHS)}): --backend hip" if amd else ""))
if len(found) > 1 or gpu_problem(found[0]) is not None:
gpu_table(found)
sel = choose_gpus(a, found) # asked when two or more cards can share the model
multi = sel if len(sel) > 1 else []
a.gpu = sel[0] # the main GPU: the checks and the sizing below are its
GPU_PICK = a.gpu
gpu = gpu_info(a.gpu)
chosen = [gpu_info(i) for i in sel]
gpu["archs"] = sorted({x["arch"] for x in chosen}) # the engine needs code for every one of them
if multi:
ok("GPUs: " + " + ".join(gpu_name(x) for x in chosen) + " together (the model's layers are split across them)")
ok(f"GPU: {gpu['name']}, {gpu['vram_gb']:.1f} GB VRAM, compute capability {cc(gpu)}, driver {gpu['driver']}")
if driver_major(gpu) < MIN_DRIVER:
fail(f"the NVIDIA driver is too old ({gpu['driver']}; {MIN_DRIVER} or newer is needed)",
"update it with the NVIDIA App or from https://www.nvidia.com/drivers, restart, and run this again")
if gpu["vram_gb"] < 11:
warn("less than 12 GB of VRAM: Strata will run, but most experts stay on the CPU and it will be slow"
+ (" - 🍷 Winery Pocket (--family winery --model POCKET) is sized for 8 GB cards"
if gpu["vram_gb"] >= SMALL_CARD_GB else ""))
ram = ram_gb()
cpu, avx2, avx512 = cpu_info()
need = min(d["ram_gb"] for d in MODELS.values())
low_ok = low_ram_fits("IQ1_M", ram, gpu["vram_gb"]) and a.low_ram != "off" # the smallest model, mapped
if ram < need - 4 and not a.check and not low_ok:
# every model keeps ALL its experts in RAM (23+ GB); VRAM only holds a copy of the most-used ones, so a
# bigger GPU does not lower this. The owner's rule: a stop by default, a risk the user can take (--model
# with --yes, or y)
confirm_risk(f"RAM: {ram:.0f} GB - the smallest model (the Coder) needs about {need} GB: Strata keeps all of "
"the model's experts in RAM (23-50 GB, whatever the GPU), so the OS will page them from disk. "
"Expect it to be very slow, and it may not start at all.", bool(a.model), a.yes,
f"RAM: {ram:.0f} GB - the smallest model (the Coder) needs about {need} GB",
"Strata keeps all of the model's experts in RAM (23-50 GB, whatever the GPU) and the GPU holds a "
"copy of the most-used ones: it needs 32 GB of RAM or more (48 GB for the full model); --model "
"NAME --yes installs one anyway")
warn(f"going on with {ram:.0f} GB of RAM, as you chose")
ok(f"RAM: {ram:.0f} GB" if ram >= need - 4 else f"RAM: {ram:.0f} GB (less than the {need} GB the smallest model needs)"
+ ("; the GPU's VRAM makes up for it (the low-RAM mode)" if ram < need - 4 and low_ok else ""))
pf = page_file_gb()
if pf is not None and pf < 4:
warn(f"Windows' page file is {pf:.1f} GB: the graphics card's memory needs room there too (issue #60), so "
"the model may not start or may use less VRAM. Set it to \"System managed\": System > About > "
"Advanced system settings > Performance > Advanced > Virtual memory")
ok(f"CPU: {cpu} ({'AVX-512' if avx512 else 'AVX2' if avx2 else 'no AVX2'})")
if not avx2:
fail("this CPU has no AVX2; Strata needs at least AVX2")
if a.check:
say()
for m, d in MODELS.items():
verdict = "fits" if ram >= d["ram_gb"] else "tight" if ram >= d["ram_gb"] - 8 else "does not fit"
if d.get("budget"):
verdict = (f"EXPERIMENTAL, fits with {resident_budget_gib(m, ram)} GiB of its experts in RAM, the rest "
"read from the SSD" if ram >= d["ram_gb"] else "does not fit")
if hip: # #429: not run on AMD yet (its prompt kernels are CUDA-only)
verdict += " - NVIDIA only so far, untested on AMD"
elif low_ram_needed(m, ram) and low_ram_fits(m, ram, gpu["vram_gb"]) and a.low_ram != "off":
verdict = (f"fits in the low-RAM mode (the GPU holds ~{100 * low_ram_gpu_share(m, gpu['vram_gb']):.0f}% "
"of its experts, " + ("the rest stays in RAM)" if low_ram_resident(m, ram, gpu["vram_gb"])
else "the rest is read from the SSD as needed)"))
say(f" {m:8s} needs ~{d['ram_gb']} GB RAM: {verdict}")
say("\nThis PC can run Strata. Run it again without --check to install.")
return 0
# ---- 2. the questions
step(2, "your choices")
fams = list(FAMILIES)
if a.family:
family = a.family
else:
for i, f in enumerate(fams, 1):
d = FAMILIES[f]
say(f" {i}) {d['title']:20s} {d['by']} - {d['about']}" + (" [experimental]" if d.get("experimental") else ""))
say(f" {len(fams) + 1}) {'Winery 9B cellar':20s} WineryLabs' dense Qwen3.5-9B blends - Grand Cru, Fable "
"Atelier, Fable Reserve, Atelier, Assemblage (Q8_0, 9.5 GB, llama.cpp)")
pick = int(ask("Which model?", [str(i) for i in range(1, len(fams) + 2)], "1", a.yes))
if pick == len(fams) + 1:
return cellar.run(a, sys.modules[__name__])
family = fams[pick - 1]
fam = FAMILIES[family]
ok(f"model: {fam['title']}")
if fam.get("license"):
say(f" Its license: {fam['license']}")
say()
names = [m for m in MODELS if family in MODELS[m].get("families", ("qwen", "swift"))]
if a.model and a.model not in names:
# #444: say which family has that size, and (with --gguf-dir) which files Strata can run at all
elsewhere_fams = [f for f in FAMILIES if f in MODELS[a.model].get("families", ("qwen", "swift"))]
fail(f"{fam['title']} has no {a.model} model file", "choose one of: " + ", ".join(names)
+ (f" (or {a.model}: " + ", ".join(f"--family {f} --model {a.model}" for f in elsewhere_fams) + ")"
if elsewhere_fams else "")
+ (f".\n {SUPPORTED_GGUFS}" if a.gguf_dir else ""))
for i, m in enumerate(names, 1):
d = MODELS[m]
fit = "" if ram >= d["ram_gb"] else f" <- needs {d['ram_gb']} GB RAM, you have {ram:.0f}"
if d.get("budget"):
say(f" {i}) {m} {d['about']}; download {d['download_gb']:.0f} GB, keeps ~"
f"{resident_budget_gib(m, ram)} GB of its {d['arena_gb']:.0f} GB of experts in RAM{fit}")
continue
if low_ram_needed(m, ram) and low_ram_fits(m, ram, gpu["vram_gb"]) and a.low_ram != "off":
fit = (f" <- fits in the low-RAM mode (the GPU holds ~{100 * low_ram_gpu_share(m, gpu['vram_gb']):.0f}%, "
+ ("the rest in RAM)" if low_ram_resident(m, ram, gpu["vram_gb"]) else "the rest from the SSD)"))
say(f" {i}) {m:8s} {d['about']}; download {d['download_gb']:.0f} GB, uses ~{d['arena_gb']:.0f} GB of RAM{fit}")
rec = str(names.index("IQ3_XXS") + 1) if ram >= 60 and "IQ3_XXS" in names else "1"
if family == "winery": # 🍷 the biggest tier this PC's RAM holds
rec = str(names.index("POCKET" if gpu["vram_gb"] < POCKET_VRAM_GB or ram < 40 else
"GRAND" if ram >= 120 else "RESERVE" if ram >= 60 else "SLIM") + 1)
model = a.model or names[int(ask("Which size?", [str(i) for i in range(1, len(names) + 1)], rec, a.yes)) - 1]
budget, q4_split = None, False
if MODELS[model].get("budget"):
# Unsloth's UD-Q4_K_XL: a RAM budget of experts, the rest from the GGUF on the SSD - not the low-RAM mode (no
# experts.bin: it would be another 77 GB on the disk), and one GPU (the budget mode has no layer split) unless
# the RAM holds the GGUFs and 24 GB more: then several, without the budget, if asked for (#498)
if family == "winery" and resident_budget_gib(model, ram) >= MODELS[model]["arena_gb"] / 1.073741824 - 0.5:
ok(f"{model}: all {MODELS[model]['arena_gb']:.0f} GB of experts fit in this PC's {ram:.0f} GB of RAM - "
"nothing is read from the SSD while it answers")
else:
warn(f"{model} is EXPERIMENTAL (docs/UNSLOTH_Q4.md): most of its experts are read from the SSD while it "
"answers, so it is several times slower than the 2-3-bit models; quality checked against llama.cpp")
if hip:
# #429 (jkuepker): checked before the 111 GB download. The HIP engine has no prompt kernels for its
# Q4_K / Q5_K experts (STRATA_MMQ_KQUANTS is CUDA-only) and it has not been run on AMD: asked, not refused
confirm_risk(f"{model} has not been run on AMD cards yet: its prompt kernels are NVIDIA-only, so on "
f"{gpu_name(gpu)} long prompts read much more slowly, and it may not work at all",
bool(a.model), a.yes, f"{model} is NVIDIA-only so far", "choose one of the 2-3-bit models, "
f"or --model {model} --yes to try it on AMD anyway", " Try it anyway?")
warn(f"installing {model} on an AMD card, as you chose (please report how it runs)")
if ram < MODELS[model]["ram_gb"]:
confirm_risk(f"{model} needs {MODELS[model]['ram_gb']} GB of RAM or more; this PC has {ram:.0f} GB: "
f"its RAM budget would be {resident_budget_gib(model, ram)} GiB, so nearly every expert is "
"read from the SSD while it answers (very slow), and it may run out of RAM",
bool(a.model), a.yes, f"{model} needs {MODELS[model]['ram_gb']} GB of RAM or more; this PC "
f"has {ram:.0f} GB", f"choose one of the 2-3-bit models, or --model {model} --yes to "
"install it anyway", " Install it anyway?")
warn(f"installing {model} with {ram:.0f} GB of RAM, as you chose")
budget = budget_choice(model, ram, a.resident_budget_gib)
if multi and not unsloth_together(a, model, ram, gpu, chosen):
multi, sel, chosen = [], [gpu["index"]], [gpu]
q4_split = bool(multi) # #498: on several GPUs without the RAM budget
if not q4_split:
ok(f"RAM budget: {budget:g} GiB of {model}'s experts in RAM, the rest read from the model files on the SSD")
if a.low_ram not in ("auto", "off"):
warn(f"--low-ram {a.low_ram} does not apply to {model}: it always reads part of its experts from the files")
elif a.resident_budget_gib is not None:
warn(f"--resident-budget-gib is for UD-Q4_K_XL: {model} keeps all of its experts in RAM or in the low-RAM mode")
low_ram = budget is None and (a.low_ram in ("on", "resident", "mmap") or
(a.low_ram == "auto" and low_ram_needed(model, ram)))
if low_ram and multi and not low_ram_together(a, model, ram, gpu, chosen):
multi, sel, chosen = [], [gpu["index"]], [gpu]
# (the low-RAM mode's variant is decided once the context is known, below; on several GPUs it is the mapped one)
if not low_ram and budget is None and ram < MODELS[model]["ram_gb"] - 4:
confirm_paging(model, ram, a.low_ram, a.yes, bool(a.model))
ok(f"size: {model}")
tag = fam["tag"] + model # names of the pack, config and start script
small = min(x["vram_gb"] for x in chosen) # each card keeps its layers' KV of the whole context
rec_ctx = 32768 if small < 14 else 65536 if small < 20 else 131072
pocket_card = model == "POCKET" and small < 14 # 🍷 64K with a 4-bit KV costs a small card what 32K at 8-bit does
if pocket_card:
rec_ctx = 65536
if budget is not None: # UD-Q4_K_XL: every GB of KV is a GB fewer of cached experts
rec_ctx = 8192 if small < 14 else 32768
# #406: the RAM rule is part of the recommendation (the smaller of the two), no longer a cap over the user's choice
rec_ctx = min(rec_ctx, ram_ctx(model, ram, low_ram))
if a.context:
ctx = a.context
else:
say()
say(" Context length = how much text the model can see at once (your chat, files, tool output).")
say(" Longer needs more VRAM for it, so fewer experts fit on the GPU:")
for i, c in enumerate(CONTEXTS, 1):
need_c = ctx_ram_need(model, c, low_ram)
note = (" (recommended for your GPU)" if c == rec_ctx else "") + \
(" (experimental: setup adds rope scaling)" if c > 262144 else "") + \
(f" (needs ~{need_c:.0f} GB RAM, this PC has {ram:.0f}: may run out of memory)"
if c > 131072 and need_c is not None and need_c > ram else "")
say(f" {i}) {c // 1024}K tokens{note}")
ctx = CONTEXTS[int(ask("Context?", [str(i) for i in range(1, len(CONTEXTS) + 1)],
str(CONTEXTS.index(rec_ctx) + 1), a.yes)) - 1]
# #406 #364: a context past the RAM rule (an explicit --context, a pick in the list, or the earlier install's) is
# kept, with what it risks. It used to become 128K: users ran 256K fine where setup's estimate said no.
need_gb = ctx_ram_need(model, ctx, low_ram)
if need_gb is not None and ram < need_gb and ctx > 131072:
warn(f"{ctx // 1024}K with {model} needs ~{need_gb:.0f} GB of RAM by setup's estimate "
f"({MODELS[model]['arena_gb']:.0f} GB of experts + the context + room for the rest); this PC has "
f"{ram:.0f}. Kept as you chose: it may be slower or run out of RAM under load. {rec_ctx // 1024}K is the "
"recommended size.")
scaling = a.rope_scaling
if ctx > 262144 and scaling is None and not a.yes:
# the interactive path: one question, yarn preselected (llama.cpp's extension method, recall-tested
# here at 512K). With --yes nothing prints: resolve_rope takes yarn below and the ok() line says so.
say()
say(f" A {ctx // 1024}K context runs the model past its trained 262,144 positions: the rotary angles")
say(" get rescaled (llama.cpp's RoPE extension). yarn keeps the trained angles on the high-frequency")
say(" pairs and corrects the magnitudes; linear shrinks every angle. Override any time with")
say(" --rope-scaling.")
scaling = ask("RoPE extension method?", ["yarn", "linear"], "yarn", a.yes)
try:
scaling, rope_scale = resolve_rope(ctx, scaling, a.rope_scale)
except ValueError as e:
fail(str(e))
if scaling is not None:
origin = ("final context / trained 262144; override with --rope-scale" if a.rope_scale is None
else "as requested")
ok(f"rope scaling: {scaling}, factor {rope_scale:g} ({origin})")
ok(f"context: {ctx} tokens")
# the KV cache (the model's memory of the conversation): 8-bit, or 4-bit after a Hadamard rotation (PR #21)
kv = "fp16" if ctx <= 8192 else (a.kv or ("q4_0" if pocket_card else "int8"))
if ctx > 8192 and not a.kv and not a.yes:
say()
say(" KV cache precision (the model's memory of the conversation):")
say(" 1) 8-bit " + ("what every published number was measured with" if pocket_card else
"(recommended: what every published number was measured with)"))
say(" 2) 4-bit half the memory (about 4% faster at 128K), but measurably less precise on long")
say(" documents; long-context lookups (needle tests) still pass"
+ (" (recommended for this GPU: more VRAM for the experts)" if pocket_card else ""))
kv = ["int8", "q4_0"][int(ask("KV cache?", ["1", "2"], "2" if pocket_card else "1", a.yes)) - 1]
if ctx > 8192:
ok(f"KV cache: {'8-bit' if kv == 'int8' else '4-bit (Hadamard-rotated)'}")
if fam.get("vision") is False:
vision = "none"
if a.vision not in (None, "no", "none"):
warn(f"images are not available with {model} yet: off")
elif hip:
vision = hip_vision(a.vision)
elif a.vision:
vision = {"yes": "gpu", "no": "none"}.get(a.vision, a.vision)
else:
say()
say(" Images: the model can also read pictures (screenshots, photos, scanned pages). This adds a 0.9 GB")
say(" download and keeps ~1.4 GB of VRAM free for the image encoder, so text is a few % slower.")
vision = "gpu" if ask("Do you want images?", ["y", "n"], "n", a.yes) == "y" else "none"
ok("images: " + {"none": "off", "gpu": "on", "cpu": "on (encoder on the CPU)"}[vision])
# The low-RAM mode's two variants. resident: the experts the GPU's cache does not hold (and, as far as RAM allows,
# the ones the prompt path borrows cache room from) are copied from the pack's experts.bin into RAM once, so
# nothing is read from the SSD while it answers (engine 0.1.30, --resident-experts; the engine falls back to mmap
# with a warning when they do not fit the RAM it finds free). mmap: they are read through the OS file cache.
# The GPU's share: its VRAM less the dense weights and buffers, this context's KV cache and the image encoder's room.
resident = False
if low_ram:
arena = MODELS[model]["arena_gb"]
vram = gpu["vram_gb"] - (VISION[vision]["reserve_mib"] / 1024 if vision != "none" else 0)
share = low_ram_gpu_share(model, vram, ctx, kv)
rest = arena - low_ram_gpu_gb(model, vram, ctx, kv)
resident = a.low_ram == "resident" or (a.low_ram != "mmap" and low_ram_resident(model, ram, vram, ctx, kv))
if multi: # #364 #384: every chosen card's share (the image encoder on the main one), the mapped variant
held = min(arena, low_ram_gpu_gb(model, vram, ctx, kv) +
sum(low_ram_gpu_gb(model, x["vram_gb"], ctx, kv) for x in chosen[1:]))
share, resident = held / arena, False
ok(f"low-RAM mode on {len(chosen)} GPUs: {model}'s experts ({arena:.0f} GB) are read from the model folder "
f"through the OS file cache instead of a copy in RAM ({ram:.0f} GB); the GPUs hold ~{100 * share:.0f}% "
"of them")
if share < 0.6:
warn("most of the experts are read from the SSD while it answers: expect it to be much slower than "
"with enough RAM (a faster SSD and a smaller size help)")
elif resident:
ok(f"low-RAM mode: the GPU holds ~{100 * share:.0f}% of {model}'s experts ({arena:.0f} GB) and the other "
f"~{rest:.0f} GB stay in RAM ({ram:.0f} GB), read once from a copy in the model folder")
else:
ok(f"low-RAM mode: {model}'s experts ({arena:.0f} GB) are read from the model folder through the OS file "
f"cache instead of a copy in RAM ({ram:.0f} GB); the GPU holds ~{100 * share:.0f}% of them")
if share < 0.6:
warn("most of the experts are read from the SSD while it answers: expect it to be much slower than "
"with enough RAM (a faster SSD and a smaller size help)")
# EXPERIMENTAL: the experimental-speed-projection control vector (data/experimental-speed-projection), off unless
# chosen here; with it loaded, the web app and the API switch it off per request
esp = None
esp_choice = (a.experimental_speed_projection or "").strip()
if family in ("qwen", "coder"): # the Coder: the same model's residual stream
if not esp_choice:
say()
say(" EXPERIMENTAL - speed projection: a small control vector applied while the model runs (layers 4-44).")
say(" It changes how the model answers: its package describes it as a refusal-direction projection (the")
say(" model declines far fewer requests). Off unless you choose it; when on, the web app can switch it off")
say(" per chat. Details: data/experimental-speed-projection/README.md")
esp_choice = "on" if ask("Turn on the experimental speed projection?", ["y", "n"], "n", a.yes) == "y" else "off"
if esp_choice.lower() not in ("off", "no", "n", "0"):
esp = ESP_VECTOR if esp_choice.lower() in ("on", "yes", "y", "1") else Path(esp_choice).expanduser().resolve()
if not esp.is_file():
fail(f"the experimental speed projection's vector is missing: {esp}")
ok("experimental speed projection: " + ("ON (experimental)" if esp else "off"))
elif esp_choice.lower() not in ("", "off", "no", "n", "0"):
warn("the experimental speed projection is made for the original Qwen3.8-Flash-Next, not Swift 1.5: left off"
if family == "swift" else f"the experimental speed projection is not tested with {model}: left off")
models_dir = Path(a.gguf_dir) if a.gguf_dir else Path(a.models_dir) / tag
shards = gguf_dir_shards(models_dir, fam, model) if a.gguf_dir else \
[models_dir / shard_file(fam, model, i) for i in range(1, n_shards(fam, model) + 1)]
problem = gguf_dir_problem(models_dir, shards[0], fam, model) if a.gguf_dir else None
if problem: # #444: files Strata cannot run, or another choice's files
fail(*problem)
if not a.gguf_dir and not all(sh.exists() and done(sh) for sh in shards):
for r in elsewhere: # already downloaded in a Strata folder on another drive
cand = [r / "models" / tag / sh.name for sh in shards]
if all(c.exists() and done(c) for c in cand):
models_dir, shards = cand[0].parent, cand
ok(f"model files found in {models_dir}")
break
for s in shards: # #173: a whole file copied in by hand has no finish mark
if s.exists() and not done(s) and whole_shard(s):
mark(s, "whole (checked against its own tensor directory)")
have_model = all(s.exists() and (done(s) or a.gguf_dir) for s in shards)
# #425 (jctaborda): a download that resumes needs room only for what is still missing - the finished shards and
# the .part files already on the disk count
on_disk = sum(f.stat().st_size for s in shards for f in (s, s.with_name(s.name + ".part")) if f.is_file()) / 1e9
to_fetch = 0 if a.gguf_dir or have_model else max(MODELS[model]["download_gb"] - on_disk, 0)
need = to_fetch + 8 + \
(40 if model == "Q2_0" and avx512 and family == "qwen" else 0) + (1 if vision != "none" else 0) + \
(MODELS[model]["arena_gb"] + 1 if low_ram and not (model == "Q2_0" and avx512 and family == "qwen") else 0)
if free_gb(models_dir) < need:
fail(f"not enough free disk space in {models_dir}: need ~{need:.0f} GB" +
(f" ({on_disk:.0f} GB of the model is already there)" if on_disk >= 1 and not have_model else ""),
"use --models-dir on a bigger drive")
# ---- 3. python packages
step(3, "Python packages")
pip_install(requirement_lines() if REQUIREMENTS.exists() else PY_PACKAGES,
"numpy, jinja2, regex, pyyaml, tqdm, requests, cmake, ninja, pillow, psutil")
# ---- 4. the engine
step(4, "the Strata engine")
llama = get_llama_cpp()
ok(f"llama.cpp {LLAMA_CPP_COMMIT[:7]} (gguf-py, ggml, mtmd)")
if hip and WIN: # AMD on Windows: the ready-made HIP engine (no compiler)
eng = None if a.build else get_prebuilt_hip(a.prebuilt, gpu)
if eng is None:
fail("no ready-made AMD engine for this Strata version" + (" (--build)" if a.build else ""),
"compiling it on Windows: tools\\hip\\build_windows.bat makes strata-windows-x64-hip.zip, then run "
"START-HERE.bat --backend hip --prebuilt <its dist folder> (docs/AMD_HIP.md)")
gpu = hip_card(eng, gpu, amd)
a.gpu = gpu["index"] if gpu["count"] > 1 else a.gpu
else:
eng = None if a.build or hip else get_prebuilt(a.prebuilt, gpu, vision)
if eng is not None and not hip and json.loads((eng / "BUILD.json").read_text()).get("source") != "local":
pip_install(CUDA_WHEELS, "NVIDIA CUDA libraries (cuBLAS, CUDA runtime; ~0.4 GB)")
if vision != "none" and not (eng / VEXE).exists():
warn("the ready-made engine has no image encoder: compiling it")
eng = None
else:
vision = prebuilt_vision(json.loads((eng / "BUILD.json").read_text()), gpu, vision)
if eng is None:
eng = build_engine_hip(gpu, llama, vision) if hip else build_engine(gpu, vision, a.yes, llama)
meta = json.loads((eng / "BUILD.json").read_text())
if hip and WIN: # the ready-made engine's rocm/bin, first on the engine's PATH
lib_dirs = [str(d) for d in hip_lib_dirs(eng)]
else:
lib_dirs = meta.get("lib_dirs") or meta.get("cuda_dirs") or cuda_lib_dirs()
engine_ver = tuple(int(x) for x in str(meta.get("version", "0")).split(".")[:3] if x.isdigit())
if budget is not None and engine_ver < UNSLOTH_ENGINE: # checked before the 111 GB download
fail(f"{model} needs engine {'.'.join(map(str, UNSLOTH_ENGINE))} or newer; this one is {meta.get('version')}",
"update Strata (or compile the engine with --build) and run setup again")
ok(f"engine: {eng / EXE}")
# ---- 5. the model files
step(5, f"downloading {fam['title']} {model}")
if not a.gguf_dir:
missing = [s.name for s in shards if not (s.exists() and done(s))]
if missing: # #495: files downloaded by hand go here, or --gguf-dir
say(f" The model files go in {models_dir}")
say(f" Files you already have: put them here with their original names ({', '.join(missing)}), or use "
"--gguf-dir <their folder>.")
if hf_endpoint() != HF_DEFAULT:
say(f" Downloading from {hf_endpoint()} (HF_ENDPOINT)")
for s in shards:
if s.exists() and done(s):
ok(f"{s.name} already downloaded")
continue
# the original's shard 2 is the same file for all its sizes and the Coder: reuse one that is already here
other = [p for p in Path(a.models_dir).glob("*/Qwen3.8-Flash-Next-GSQ-RCO-*-00002-of-00002.gguf") if done(p)]
if family in ("qwen", "coder") and s.name.endswith("00002-of-00002.gguf") and other and not s.exists():
try:
os.link(other[0], s)
mark(s)
ok(f"{s.name} shared with {other[0].parent.name} (identical file)")
continue
except OSError:
pass
download(fam["hf"].format(q=model) + s.name, s)
check_shards(shards)
for s in shards: # the experimental Unsloth file: pinned sizes and SHA-256
if s.name in fam.get("sha256", {}):
verify_sha256(s, *fam["sha256"][s.name])
ok("model files present")
mmproj = Path(a.models_dir) / fam["mmproj"]
if not mmproj.exists():
mmproj = find_in(roots, f"models/{fam['mmproj']}") or mmproj
if vision != "none":
if not mmproj.exists() and a.gguf_dir and (Path(a.gguf_dir) / fam["mmproj"]).exists():
mmproj = Path(a.gguf_dir) / fam["mmproj"]
else:
download(fam["mmproj_hf"] + fam["mmproj"], mmproj, "vision encoder")
ok(f"vision encoder: {mmproj}")
# ---- 6. the pack and the MTP draft layer
step(6, "preparing the model for Strata")
pack = find_in(roots, f"packs/{tag.lower()}") or data / "packs" / tag.lower()
env = dict(os.environ, STRATA_GGUF_PY=str(llama / "gguf-py"))
if model == "Q2_0" and avx512 and family == "qwen":
# the Q2_0 experts repacked for the AVX-512 kernel (the measured speed): a one-time ~40 GB conversion
if not (pack / "index.txt").exists() or not (pack / "experts.bin").exists(): # index.txt is written last
say(" Converting the Q2_0 experts for the AVX-512 kernel (one time, ~40 GB written, 2-5 min) ...")
run([sys.executable, str(ROOT / "tools" / "strata_pack.py"), "build", "--gguf", str(shards[0]),
"--out", str(pack), "--skip-hash"], env=env)
run([sys.executable, str(ROOT / "tools" / "pack_index.py"), "--pack", str(pack)], env=env)
if not (pack / "tokenizer" / "vocab.json").exists():
run([sys.executable, str(ROOT / "tools" / "strata_tokenizer.py"), "--gguf", str(shards[0]),
"--out", str(pack)], env=env) # writes <pack>/tokenizer/
elif not (pack / "native_experts.txt").exists() or not (pack / "tokenizer" / "vocab.json").exists():
# every tensor as the GGUF stores it; the experts are read from the GGUF at start (seconds to build)
# (UD-Q4_K_XL: --compat-bf16 - its Q8_0 hyper-connection projections become BF16, the form the engine reads)
run([sys.executable, str(ROOT / "tools" / "iq_pack.py"), "--gguf", str(shards[0]), "--out", str(pack),
*fam.get("pack_args", [])], env=env)
if low_ram and not (pack / "experts.bin").exists():
say(f" Writing the experts into one file for the low-RAM mode (one time, {MODELS[model]['arena_gb']:.0f} GB) ...")
run([sys.executable, str(ROOT / "tools" / "iq_pack.py"), "--gguf", str(shards[0]), "--out", str(pack),
"--experts-bin"], env=env)
ok(f"model prepared: {pack}")
mtp = (find_in(roots, "mtp/rt/experts.bin") or data / "mtp/rt/experts.bin").parent.parent
rt = mtp / "rt"
corrupt = (rt / "experts.bin").exists() and mtp_corrupt(mtp, env)
if corrupt:
warn("some MTP tensors are not the checkpoint's (a download mirror that ignored range requests, #327): "
"fetching them again and rebuilding the draft layer")
if corrupt or not (rt / "experts.bin").exists():
say(" The MTP draft layer (speculative decoding, ~2x faster output) comes from the original Qwen checkpoint:")
say(" only its ~5 GB of MTP tensors are downloaded.")
run([sys.executable, str(ROOT / "tools" / "mtp_fetch.py"), "fetch", "--out", str(mtp)], env=env)
run([sys.executable, str(ROOT / "tools" / "mtp_pack.py"), "--src", str(mtp), "--experts", "q2_0",
"--out", str(mtp / "mtp-q2_0.gguf")], env=env)
run([sys.executable, str(ROOT / "tools" / "mtp_rt.py"), "--gguf", str(mtp / "mtp-q2_0.gguf"), "--out", str(rt)],
env=env)
# a setup run again without --draft-vocab keeps the subset this model's config chose before (cyrillic, en)
draft_vocab = a.draft_vocab or saved_draft_vocab(ROOT / f"strata-{tag.lower()}.json")
refresh_draft_vocab(rt, draft_vocab or "cjk")
ok(f"MTP draft layer: {rt}")
for line in draft_vocab_note(gpu.get("vram_gb", 0.0), draft_vocab): # #474: a recommendation, nothing changes
say(" " + line)
# ---- 7. the start script
step(7, "writing the start script")
sys.path.insert(0, str(ROOT / "tools"))
from gguf_reader import GGUFFile # the PLE table's shard: shard 2 (original) or 1 (Swift)
ple = next((s for s in shards if any(t.name == "per_layer_token_embd.weight" for t in GGUFFile(s).tensors)), None)
if ple is None:
fail("the model has no per_layer_token_embd tensor (is this a Qwen3.8-Flash-Next GGUF?)")
# (a 4-shard file: the engine finds the PLE table's shard itself from shard 1, the measured setup)
args = ["--pack", str(pack), "--native", str(shards[0]), *(["--ple-gguf", str(ple)] if len(shards) <= 2 else []),
"--expert-profile", str(ROOT / "data" / MODELS[model].get("profile", fam.get("profile", "expert-profile.bin"))), "--expert-cache", "auto",
"--prefill", "auto", "--spec", "4", "--spec-min-p", "0.5", "--mtp", str(rt),
"--max-context", str(ctx)]
if scaling is not None: # the resolved config: explicit flags as given, or the automatic yarn+factor
args += ["--rope-scaling", scaling, "--rope-scale", f"{rope_scale:g}"]
if ctx > 8192:
args += ["--kv", kv]
if resident and a.low_ram != "resident" and engine_ver < RESIDENT_ENGINE:
resident = False # an engine from before --resident-experts would refuse it
ok(f"low-RAM mode: engine {meta.get('version')} has no resident variant yet; the experts are read through "
"the OS file cache (run setup again after the next engine update)")
if low_ram: # the experts from the pack's experts.bin: the ones the GPU does not hold copied into RAM, or mapped
args += ["--resident-experts" if resident else "--mmap-experts"]
# KV streaming: from 64K up the whole KV cache lives in RAM and only the part the attention reads (32K positions
# per layer) stays in VRAM; the VRAM it frees holds more experts (+6% at 128K, +23% at 262K with Q2_0). It
# costs ~13.7 KB of RAM per context token with 8-bit KV (1.7 GB at 128K), 7.5 KB with 4-bit, so only when it fits.
kv_ram_gb = ctx * (13 * (576 if kv == "q4_0" else 1056)) / 1e9 # 12 QSA layers + the draft layer
# Hybrid K8V4 never streams its KV (mode 0 only, layer.hpp), so it is excluded from the WHOLE streaming
# decision rather than one threshold at a time - a future tier added to this chain cannot reintroduce the
# combination the engine refuses (PR review).
# --kv-streaming on|off overrides the RAM test (the owner's rule); k8v4 and WSL stay off - they cannot stream.
stream_fits = ram >= MODELS[model]["ram_gb"] + kv_ram_gb + 1
if kv == "k8v4":
if ctx >= 65536:
ok("KV streaming off: not supported with --kv k8v4; the KV cache stays in VRAM")
if a.kv_streaming == "on":
warn("--kv-streaming on: the engine has no KV streaming with --kv k8v4 (it refuses the pair): off")
elif is_wsl() and ctx >= 65536:
ok("WSL: KV streaming off (the driver pins only about 1 GB of RAM); the KV cache stays in VRAM")
if a.kv_streaming == "on":
warn("--kv-streaming on: WSL cannot stream the KV cache (its RAM copy must be pinned, and the driver pins "
"only about 1 GB there): off")
elif ctx >= 65536 and a.kv_streaming == "off":
ok("KV streaming off, as you chose (--kv-streaming off): the KV cache stays in VRAM")
elif ctx >= 65536 and (stream_fits or a.kv_streaming == "on"):
args += ["--kv-resident", "32768"]
ok(f"KV streaming on: the context's KV cache lives in RAM ({kv_ram_gb:.1f} GB), more experts fit in VRAM")
if not stream_fits:
warn(f"KV streaming needs ~{kv_ram_gb:.1f} GB of RAM beside the ~{MODELS[model]['ram_gb']} GB {model} "
f"uses; this PC has {ram:.0f}. Kept as you chose (--kv-streaming on): it may page or run out of RAM "
"under load")
if q4_split: # #498: no budget to take it out of
pass
elif budget is not None and a.resident_budget_gib is None: # its RAM comes out of the experts' budget
budget = resident_budget_gib(model, ram, kv_ram_gb)
ok(f"RAM budget: {budget} GiB (less the KV cache's RAM)")
elif budget is not None and budget > resident_budget_gib(model, ram, kv_ram_gb):
warn(f"the KV cache's {kv_ram_gb:.1f} GB of RAM come on top of your {budget:g} GiB RAM budget (setup "
f"would take them out of it: {resident_budget_gib(model, ram, kv_ram_gb)} GiB); kept as you chose")
elif a.kv_streaming == "on":
warn("--kv-streaming on: a context under 64K is not streamed (the attention's window holds all of it): off")
if budget is not None and not q4_split: # UD-Q4_K_XL: the experts read from the GGUF in place, the most-used N
args += ["--resident-budget-gib", f"{budget:g}"] # GiB kept in RAM (#498: a layer split has no budget)
if vision != "none":
args += ["--vision", "--vram-reserve-mib", str(VISION[vision]["reserve_mib"])]
if a.vram_reserve_mib is not None: # #493: VRAM left free for other programs (only when given)
if "--vram-reserve-mib" in args:
i = args.index("--vram-reserve-mib") + 1
if vision == "gpu" and a.vram_reserve_mib < int(args[i]):
warn(f"--vram-reserve-mib {a.vram_reserve_mib}: the image encoder on the GPU needs ~{args[i]} MiB of "
"it; kept as you chose (it may run out of VRAM when it reads a picture)")
args[i] = str(a.vram_reserve_mib)
else:
args += ["--vram-reserve-mib", str(a.vram_reserve_mib)]
ok(f"VRAM kept free for other programs: {a.vram_reserve_mib} MiB (--vram-reserve-mib; the expert cache takes "
"that much less)")
if not multi and 0 < gpu.get("vram_gb", 0.0) < SMALL_CARD_GB:
# #496: on a 6 GB card the expert cache can get no room at all; the engine lowers its own reserve when that
# is what it takes, and says what is short when even that is not enough. Setup only says what helps.
for line in small_card_note(ctx, draft_vocab): # a recommendation: nothing changes
say(" " + line)
elif hip and a.vram_reserve_mib is None and linux_desktop():
for line in desktop_reserve_note(): # #560 #516: a recommendation: nothing changes
say(" " + line)
if esp is not None:
# the package's profile, with llama.cpp's flags (the engine takes the same ones)
args += ["--control-vector-scaled", f"{esp}:1.0", "--control-vector-layer-range", "4", "44",
"--cvec-mode", "project", "--cvec-dir", "per-layer"]
cfg = {"exe": str(eng / EXE), "args": args, "cwd": str(ROOT), "tokenizer": str(pack / "tokenizer"),
"model_name": f"{fam['name']}-{model.lower()}", "log": str(ROOT / f"strata-{tag.lower()}.log"),
"lib_dirs": lib_dirs, "port": port}
if hip:
cfg["backend"] = "hip"
# the dense prompt GEMMs through hipBLASLt with kernels measured on this GPU generation (tools/hip; +40-60%
# prompt speed on the 7900 XTX): only a table for this card's arch AND the installed hipBLASLt version (the
# engine refuses any other one and falls back to plain hipBLAS)
table = hipblaslt_table(gpu["arch"], lib_dirs, meta.get("hipblaslt_version"))
if table:
cfg["env"] = {"STRATA_HIPBLASLT_TUNING": str(table)}
if resident: # ROCm: large page-locked host allocations can fail or be slow for the CPU; keep the copy pageable
cfg.setdefault("env", {})["STRATA_RESIDENT_PIN"] = "0"
if gpu["count"] > 1 or a.gpu is not None:
cfg["gpu"] = gpu["index"] # the engine is told this card (issue #51)
cfg["gpus_asked"] = True # chosen at setup: not asked again at start
if multi: # a layer split across these cards (the server adds the flag)
cfg["gpu"] = multi
cfg["layer_split"] = a.layer_split or "auto"
ok(f"layer split across GPUs {multi} ({cfg['layer_split']})")
if a.host:
cfg["host"] = a.host
if a.api_key:
cfg["api_key"] = a.api_key
if draft_vocab:
cfg["draft_vocab"] = draft_vocab
if vision != "none":
cfg["vision"] = {"exe": str(eng / VEXE), "mmproj": str(mmproj), "model": str(shards[0]),
"gpu": vision == "gpu", "max_tokens": VISION[vision]["max_tokens"]}
if vision == "cpu":
cfg["vision"]["threads"] = max(1, (os.cpu_count() or 8) // 2)
cfg_path = ROOT / f"strata-{tag.lower()}.json"
cal = None if hip else saved_calibration(cfg) # tools/calibrate.py is NVIDIA-only for now
if cal is not None:
sys.path.insert(0, str(ROOT / "tools"))
import calibrate as CAL
cfg["args"] = CAL.apply(cfg["args"], cal.get("settings") or {})
ok("the settings tuned for this PC earlier are used" + (f" ({cal['date']})" if cal.get("date") else ""))
write_config(cfg_path, cfg)
script = write_run_script(tag, cfg_path, port)
# offered only when someone answers: --yes installs and adopted earlier installs are not held up by it
if cal is None and not hip and not a.no_start and not a.yes and ask(
"Tune Strata for this PC now? It measures a few engine settings (about 5-10 minutes; the PC is busy "
"meanwhile; later: START-HERE --calibrate)", ["y", "n"], "y", a.yes) == "y":
tuned = calibrate_config(cfg_path)
else:
tuned = None # not asked for: nothing to repeat below
ok(f"start script: {script.name}")
say()
say("All set.")
say(f" API (OpenAI): http://127.0.0.1:{port}/v1 (any API key; model name: anything)")
say(f" API (Anthropic): http://127.0.0.1:{port}/v1/messages")
if a.host and a.host not in ("127.0.0.1", "localhost"):
say(f" Other devices: the server window prints this PC's address (http://<IP>:{port}/)"
+ ("" if a.api_key else " - no API key set: anyone on your network can use it"))
say(f" Next time: just run {'START-HERE.bat' if WIN else './setup.sh'} (or {script.name}) - it starts right away")
if vision != "none":
say(" Images: send them in the chat page, in chat.py (/image <path>) or over the API")
if tuned is False: # #447: a failed tuning is repeated here, not only above
say(" Tuning: FAILED (the reason is above): the default settings stay - "
f"{'START-HERE.bat' if WIN else './setup.sh'} --calibrate tries again")
if a.no_start:
return 0
return start(cfg_path, port)
if __name__ == "__main__":
try:
sys.exit(main())
except KeyboardInterrupt:
say("\nstopped.")
sys.exit(1)