import { HARDWARE, hardwareLabel } from "./hardware"; export type Support = "yes" | "no" | "partial" | "unknown"; export type Backend = "cuda" | "rocm" | "metal" | "intel" | "cpu"; export interface Model { id: string; label: string; /** billions of parameters */ params: number; layers: number; kvHeads: number; headDim: number; } export interface Gpu { id: string; label: string; /** GB of VRAM */ vram: number; backend: Backend; } export interface Goal { id: string; label: string; /** capability that must be supported for this goal */ requires?: "peft" | "serializable"; } export interface Method { id: string; name: string; bits: number[]; backends: Backend[]; onTheFly: boolean; compile: Support; peft: Support; serializable: Support; /** minutes of setup before you can load */ setupMinutes: number; blurb: string; config: string; docs: string; } /** * Every memory configuration in the Hugging Face hardware table, most-owned * first — so the cards people actually have sit at the top of the list. */ export const GPUS: Gpu[] = HARDWARE.map((h) => ({ id: h.id, label: hardwareLabel(h), vram: h.vram, backend: h.backend, })); export const GOALS: Goal[] = [ { id: "serve", label: "serve requests" }, { id: "finetune", label: "fine-tune it", requires: "peft" }, { id: "publish", label: "publish a checkpoint", requires: "serializable" }, ]; export const PRECISIONS = [8, 4, 3, 2] as const; export const PRECISION_ACCENT: Record = { 8: "#c08532", 4: "#34785c", 3: "#f54e00", 2: "#cf2d56", }; export const METHODS: Method[] = [ { id: "bitsandbytes", name: "bitsandbytes", bits: [4, 8], backends: ["cuda", "rocm", "metal", "intel", "cpu"], onTheFly: true, compile: "yes", peft: "yes", serializable: "yes", setupMinutes: 0, blurb: "Quantizes as the model loads. No calibration data, no extra pipeline.", config: "BitsAndBytesConfig(load_in_4bit=True)", docs: "https://huggingface.co/docs/transformers/main/en/quantization/bitsandbytes", }, { id: "gptqmodel", name: "GPTQModel", bits: [2, 3, 4, 8], backends: ["cuda", "rocm", "metal", "intel", "cpu"], onTheFly: false, compile: "no", peft: "yes", serializable: "yes", setupMinutes: 25, blurb: "Best accuracy available at 4-bit. Costs a calibration pass up front.", config: 'GPTQConfig(bits=4, dataset="c4")', docs: "https://huggingface.co/docs/transformers/main/en/quantization/gptq", }, { id: "awq", name: "AWQ", bits: [4], backends: ["cuda", "rocm", "intel", "cpu"], onTheFly: false, compile: "unknown", peft: "yes", serializable: "yes", setupMinutes: 20, blurb: "Activation-aware and widely deployed. Slightly behind GPTQ on accuracy.", config: "AwqConfig(bits=4)", docs: "https://huggingface.co/docs/transformers/main/en/quantization/awq", }, { id: "compressed-tensors", name: "compressed-tensors", bits: [2, 3, 4, 8], backends: ["cuda", "rocm", "intel", "cpu"], onTheFly: false, compile: "no", peft: "yes", serializable: "yes", setupMinutes: 0, blurb: "Loads checkpoints that are already quantized, and sparse as well.", config: "load a compressed-tensors checkpoint", docs: "https://huggingface.co/docs/transformers/main/en/quantization/compressed_tensors", }, { id: "torchao", name: "torchao", bits: [4, 8], backends: ["cuda", "metal", "intel", "cpu"], onTheFly: true, compile: "unknown", peft: "unknown", serializable: "partial", setupMinutes: 0, blurb: "Native to PyTorch. Three of the four capabilities are unverified.", config: 'TorchAoConfig("int4_weight_only")', docs: "https://huggingface.co/docs/transformers/main/en/quantization/torchao", }, { id: "finegrained-fp8", name: "FINEGRAINED_FP8", bits: [8], backends: ["cuda", "intel"], onTheFly: true, compile: "no", peft: "no", serializable: "yes", setupMinutes: 0, blurb: "Built into Transformers, so there is no extra dependency to pin.", config: "FineGrainedFP8Config()", docs: "https://huggingface.co/docs/transformers/main/en/quantization/finegrained_fp8", }, { id: "nvfp4", name: "NVFP4", bits: [4], backends: ["cuda"], onTheFly: true, compile: "yes", peft: "no", serializable: "no", setupMinutes: 0, blurb: "NVIDIA's 4-bit float format, run through Hub kernels. Needs a Blackwell card.", config: "see the NVFP4 docs", docs: "https://huggingface.co/docs/transformers/main/en/quantization/nvfp4", }, { id: "gguf", name: "GGUF", bits: [2, 3, 4, 8], backends: ["cuda", "metal", "intel", "cpu"], onTheFly: false, compile: "yes", peft: "no", serializable: "no", setupMinutes: 0, blurb: "The llama.cpp format. Pick a file from any of the GGUF repos on the Hub.", config: 'from_pretrained(repo, gguf_file="model-Q4_K_M.gguf")', docs: "https://huggingface.co/docs/transformers/main/en/quantization/gguf", }, { id: "auto-round", name: "AutoRound", bits: [2, 3, 4, 8], backends: ["cuda", "intel", "cpu"], onTheFly: false, compile: "no", peft: "no", serializable: "yes", setupMinutes: 0, blurb: "Intel's weight-rounding method, with checkpoints from 2 to 8 bits.", config: "load an AutoRound checkpoint", docs: "https://huggingface.co/docs/transformers/main/en/quantization/auto_round", }, { id: "quark", name: "Quark", bits: [2, 4, 8], backends: ["cuda", "rocm", "metal", "intel", "cpu"], onTheFly: false, compile: "unknown", peft: "no", serializable: "no", setupMinutes: 0, blurb: "AMD's toolkit. Loads Quark checkpoints on AMD and NVIDIA hardware.", config: "load a Quark checkpoint", docs: "https://huggingface.co/docs/transformers/main/en/quantization/quark", }, ]; /** Weight bytes, in decimal GB. */ export function weightsGB(model: Model, bits: number): number { return (model.params * bits) / 8; } /** KV cache for a given context length, in decimal GB (fp16 K and V). */ export function kvCacheGB(model: Model, contextTokens: number): number { const bytesPerToken = 2 * model.layers * model.kvHeads * model.headDim * 2; return (bytesPerToken * contextTokens) / 1e9; } /** Rough allowance for activations and CUDA workspace, in decimal GB. */ export const ACTIVATIONS_GB = 1.5; export function overheadGB(model: Model, contextTokens: number): number { return kvCacheGB(model, contextTokens) + ACTIVATIONS_GB; } export function totalGB(model: Model, bits: number, contextTokens: number): number { return weightsGB(model, bits) + overheadGB(model, contextTokens); } /** * Where the quantized weights come from: made as the model loads, or read * from a checkpoint someone already quantized. */ export type Source = "on-the-fly" | "checkpoint"; export const SOURCES: { id: Source; label: string }[] = [ { id: "on-the-fly", label: "on the fly" }, { id: "checkpoint", label: "from a checkpoint" }, ]; const BELOW_4 = ["gguf", "auto-round", "compressed-tensors", "quark"]; /** * The methods we recommend for each source and precision, best first. Anything * not listed is left out even when it technically supports the combination. */ export const RECOMMENDED: Record> = { "on-the-fly": { 8: ["finegrained-fp8", "bitsandbytes", "torchao"], 4: ["bitsandbytes", "torchao", "nvfp4"], 3: [], 2: [], }, checkpoint: { 8: ["finegrained-fp8", "bitsandbytes", "torchao", "gguf", "auto-round", "compressed-tensors", "quark"], 4: ["bitsandbytes", "torchao", "nvfp4", "gguf", "auto-round", "gptqmodel", "awq", "compressed-tensors", "quark"], 3: BELOW_4, 2: BELOW_4, }, }; export function methodsFor(bits: number, backend: Backend, goal: Goal, source: Source): Method[] { return (RECOMMENDED[source][bits] ?? []) .map((id) => METHODS.find((m) => m.id === id)!) .filter((m) => { if (!m.bits.includes(bits)) return false; if (!m.backends.includes(backend)) return false; if (goal.requires === "peft" && m.peft !== "yes") return false; if (goal.requires === "serializable" && m.serializable === "no") return false; return true; }); } export function round1(n: number): number { return Math.round(n * 10) / 10; }