Spaces:
Running on CPU Upgrade
Running on CPU Upgrade
Download lib/data.ts from stevhliu/quantization-picker: direct link, hf CLI and curl.
- Browser
- Download file 8.49 kB
-
https://huggingface.co/spaces/stevhliu/quantization-picker/resolve/main/lib/data.ts
- Command line
-
hf download hf://spaces/stevhliu/quantization-picker/lib/data.ts
-
curl -L -o data.ts https://huggingface.co/spaces/stevhliu/quantization-picker/resolve/main/lib/data.ts
8.49 kB
| import { HARDWARE, hardwareLabel } from "./hardware"; | |
| export type Support = "yes" | "no" | "partial" | "unknown"; | |
| export type Backend = "cuda" | "rocm" | "metal" | "intel" | "cpu"; | |
| export interface Model { | |
| id: string; | |
| label: string; | |
| /** billions of parameters */ | |
| params: number; | |
| layers: number; | |
| kvHeads: number; | |
| headDim: number; | |
| } | |
| export interface Gpu { | |
| id: string; | |
| label: string; | |
| /** GB of VRAM */ | |
| vram: number; | |
| backend: Backend; | |
| } | |
| export interface Goal { | |
| id: string; | |
| label: string; | |
| /** capability that must be supported for this goal */ | |
| requires?: "peft" | "serializable"; | |
| } | |
| export interface Method { | |
| id: string; | |
| name: string; | |
| bits: number[]; | |
| backends: Backend[]; | |
| onTheFly: boolean; | |
| compile: Support; | |
| peft: Support; | |
| serializable: Support; | |
| /** minutes of setup before you can load */ | |
| setupMinutes: number; | |
| blurb: string; | |
| config: string; | |
| docs: string; | |
| } | |
| /** | |
| * Every memory configuration in the Hugging Face hardware table, most-owned | |
| * first — so the cards people actually have sit at the top of the list. | |
| */ | |
| export const GPUS: Gpu[] = HARDWARE.map((h) => ({ | |
| id: h.id, | |
| label: hardwareLabel(h), | |
| vram: h.vram, | |
| backend: h.backend, | |
| })); | |
| export const GOALS: Goal[] = [ | |
| { id: "serve", label: "serve requests" }, | |
| { id: "finetune", label: "fine-tune it", requires: "peft" }, | |
| { id: "publish", label: "publish a checkpoint", requires: "serializable" }, | |
| ]; | |
| export const PRECISIONS = [8, 4, 3, 2] as const; | |
| export const PRECISION_ACCENT: Record<number, string> = { | |
| 8: "#c08532", | |
| 4: "#34785c", | |
| 3: "#f54e00", | |
| 2: "#cf2d56", | |
| }; | |
| export const METHODS: Method[] = [ | |
| { | |
| id: "bitsandbytes", | |
| name: "bitsandbytes", | |
| bits: [4, 8], | |
| backends: ["cuda", "rocm", "metal", "intel", "cpu"], | |
| onTheFly: true, | |
| compile: "yes", | |
| peft: "yes", | |
| serializable: "yes", | |
| setupMinutes: 0, | |
| blurb: "Quantizes as the model loads. No calibration data, no extra pipeline.", | |
| config: "BitsAndBytesConfig(load_in_4bit=True)", | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/bitsandbytes", | |
| }, | |
| { | |
| id: "gptqmodel", | |
| name: "GPTQModel", | |
| bits: [2, 3, 4, 8], | |
| backends: ["cuda", "rocm", "metal", "intel", "cpu"], | |
| onTheFly: false, | |
| compile: "no", | |
| peft: "yes", | |
| serializable: "yes", | |
| setupMinutes: 25, | |
| blurb: "Best accuracy available at 4-bit. Costs a calibration pass up front.", | |
| config: 'GPTQConfig(bits=4, dataset="c4")', | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/gptq", | |
| }, | |
| { | |
| id: "awq", | |
| name: "AWQ", | |
| bits: [4], | |
| backends: ["cuda", "rocm", "intel", "cpu"], | |
| onTheFly: false, | |
| compile: "unknown", | |
| peft: "yes", | |
| serializable: "yes", | |
| setupMinutes: 20, | |
| blurb: "Activation-aware and widely deployed. Slightly behind GPTQ on accuracy.", | |
| config: "AwqConfig(bits=4)", | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/awq", | |
| }, | |
| { | |
| id: "compressed-tensors", | |
| name: "compressed-tensors", | |
| bits: [2, 3, 4, 8], | |
| backends: ["cuda", "rocm", "intel", "cpu"], | |
| onTheFly: false, | |
| compile: "no", | |
| peft: "yes", | |
| serializable: "yes", | |
| setupMinutes: 0, | |
| blurb: "Loads checkpoints that are already quantized, and sparse as well.", | |
| config: "load a compressed-tensors checkpoint", | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/compressed_tensors", | |
| }, | |
| { | |
| id: "torchao", | |
| name: "torchao", | |
| bits: [4, 8], | |
| backends: ["cuda", "metal", "intel", "cpu"], | |
| onTheFly: true, | |
| compile: "unknown", | |
| peft: "unknown", | |
| serializable: "partial", | |
| setupMinutes: 0, | |
| blurb: "Native to PyTorch. Three of the four capabilities are unverified.", | |
| config: 'TorchAoConfig("int4_weight_only")', | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/torchao", | |
| }, | |
| { | |
| id: "finegrained-fp8", | |
| name: "FINEGRAINED_FP8", | |
| bits: [8], | |
| backends: ["cuda", "intel"], | |
| onTheFly: true, | |
| compile: "no", | |
| peft: "no", | |
| serializable: "yes", | |
| setupMinutes: 0, | |
| blurb: "Built into Transformers, so there is no extra dependency to pin.", | |
| config: "FineGrainedFP8Config()", | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/finegrained_fp8", | |
| }, | |
| { | |
| id: "nvfp4", | |
| name: "NVFP4", | |
| bits: [4], | |
| backends: ["cuda"], | |
| onTheFly: true, | |
| compile: "yes", | |
| peft: "no", | |
| serializable: "no", | |
| setupMinutes: 0, | |
| blurb: "NVIDIA's 4-bit float format, run through Hub kernels. Needs a Blackwell card.", | |
| config: "see the NVFP4 docs", | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/nvfp4", | |
| }, | |
| { | |
| id: "gguf", | |
| name: "GGUF", | |
| bits: [2, 3, 4, 8], | |
| backends: ["cuda", "metal", "intel", "cpu"], | |
| onTheFly: false, | |
| compile: "yes", | |
| peft: "no", | |
| serializable: "no", | |
| setupMinutes: 0, | |
| blurb: "The llama.cpp format. Pick a file from any of the GGUF repos on the Hub.", | |
| config: 'from_pretrained(repo, gguf_file="model-Q4_K_M.gguf")', | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/gguf", | |
| }, | |
| { | |
| id: "auto-round", | |
| name: "AutoRound", | |
| bits: [2, 3, 4, 8], | |
| backends: ["cuda", "intel", "cpu"], | |
| onTheFly: false, | |
| compile: "no", | |
| peft: "no", | |
| serializable: "yes", | |
| setupMinutes: 0, | |
| blurb: "Intel's weight-rounding method, with checkpoints from 2 to 8 bits.", | |
| config: "load an AutoRound checkpoint", | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/auto_round", | |
| }, | |
| { | |
| id: "quark", | |
| name: "Quark", | |
| bits: [2, 4, 8], | |
| backends: ["cuda", "rocm", "metal", "intel", "cpu"], | |
| onTheFly: false, | |
| compile: "unknown", | |
| peft: "no", | |
| serializable: "no", | |
| setupMinutes: 0, | |
| blurb: "AMD's toolkit. Loads Quark checkpoints on AMD and NVIDIA hardware.", | |
| config: "load a Quark checkpoint", | |
| docs: "https://huggingface.co/docs/transformers/main/en/quantization/quark", | |
| }, | |
| ]; | |
| /** Weight bytes, in decimal GB. */ | |
| export function weightsGB(model: Model, bits: number): number { | |
| return (model.params * bits) / 8; | |
| } | |
| /** KV cache for a given context length, in decimal GB (fp16 K and V). */ | |
| export function kvCacheGB(model: Model, contextTokens: number): number { | |
| const bytesPerToken = 2 * model.layers * model.kvHeads * model.headDim * 2; | |
| return (bytesPerToken * contextTokens) / 1e9; | |
| } | |
| /** Rough allowance for activations and CUDA workspace, in decimal GB. */ | |
| export const ACTIVATIONS_GB = 1.5; | |
| export function overheadGB(model: Model, contextTokens: number): number { | |
| return kvCacheGB(model, contextTokens) + ACTIVATIONS_GB; | |
| } | |
| export function totalGB(model: Model, bits: number, contextTokens: number): number { | |
| return weightsGB(model, bits) + overheadGB(model, contextTokens); | |
| } | |
| /** | |
| * Where the quantized weights come from: made as the model loads, or read | |
| * from a checkpoint someone already quantized. | |
| */ | |
| export type Source = "on-the-fly" | "checkpoint"; | |
| export const SOURCES: { id: Source; label: string }[] = [ | |
| { id: "on-the-fly", label: "on the fly" }, | |
| { id: "checkpoint", label: "from a checkpoint" }, | |
| ]; | |
| const BELOW_4 = ["gguf", "auto-round", "compressed-tensors", "quark"]; | |
| /** | |
| * The methods we recommend for each source and precision, best first. Anything | |
| * not listed is left out even when it technically supports the combination. | |
| */ | |
| export const RECOMMENDED: Record<Source, Record<number, string[]>> = { | |
| "on-the-fly": { | |
| 8: ["finegrained-fp8", "bitsandbytes", "torchao"], | |
| 4: ["bitsandbytes", "torchao", "nvfp4"], | |
| 3: [], | |
| 2: [], | |
| }, | |
| checkpoint: { | |
| 8: ["finegrained-fp8", "bitsandbytes", "torchao", "gguf", "auto-round", "compressed-tensors", "quark"], | |
| 4: ["bitsandbytes", "torchao", "nvfp4", "gguf", "auto-round", "gptqmodel", "awq", "compressed-tensors", "quark"], | |
| 3: BELOW_4, | |
| 2: BELOW_4, | |
| }, | |
| }; | |
| export function methodsFor(bits: number, backend: Backend, goal: Goal, source: Source): Method[] { | |
| return (RECOMMENDED[source][bits] ?? []) | |
| .map((id) => METHODS.find((m) => m.id === id)!) | |
| .filter((m) => { | |
| if (!m.bits.includes(bits)) return false; | |
| if (!m.backends.includes(backend)) return false; | |
| if (goal.requires === "peft" && m.peft !== "yes") return false; | |
| if (goal.requires === "serializable" && m.serializable === "no") return false; | |
| return true; | |
| }); | |
| } | |
| export function round1(n: number): number { | |
| return Math.round(n * 10) / 10; | |
| } | |