""" Mide el uso de VRAM (pico) de un modelo en distintas presentaciones: - transformers bf16 (19.3 GB → NO cabe en 4 GB VRAM) - transformers 4-bit (bitsandbytes NF4, ~6 GB) - GGUF Q4_K_M vía llama.cpp (llama-cli / llama-server) Modo transformers: python measure_vram.py --hf-dir [--nf4] Modo GGUF: python measure_vram.py --gguf [--n-gpu-layers 999] Reporta pico de VRAM (torch.cuda.max_memory_allocated + nvidia-smi) y footprint. """ import argparse import subprocess import time from pathlib import Path PROMPTS = [ "What is the capital of France?", "Explain the water cycle in three sentences.", "Write a haiku about winter.", ] def nvidia_mem_used_mb(): out = subprocess.run( ["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader,nounits"], capture_output=True, text=True, check=False, ).stdout.strip() try: return int(out.splitlines()[0].strip()) except (ValueError, IndexError): return -1 def measure_transformers(path, nf4=False, max_new=24): import torch from transformers import AutoModelForCausalLM, AutoTokenizer if torch.cuda.is_available(): torch.cuda.reset_peak_memory_stats() kwargs = {"torch_dtype": torch.bfloat16 if not nf4 else None} if nf4: from transformers import BitsAndBytesConfig kwargs["quantization_config"] = BitsAndBytesConfig(load_in_4bit=True, bnb_4bit_compute_dtype=torch.bfloat16) print(f"Cargando {'NF4 4-bit' if nf4 else 'bf16'} desde {path} ...") model = AutoModelForCausalLM.from_pretrained(str(path), **kwargs, device_map="auto") tokenizer = AutoTokenizer.from_pretrained(str(path)) print(f" footprint: {model.get_memory_footprint() / 1e9:.2f} GB") before = nvidia_mem_used_mb() device = next(model.parameters()).device with torch.inference_mode(): for p in PROMPTS: inputs = tokenizer(p, return_tensors="pt").to(device) out = model.generate(**inputs, max_new_tokens=max_new, do_sample=False) print(f" → {tokenizer.decode(out[0][inputs['input_ids'].shape[1]:], skip_special_tokens=True)[:60]}") if torch.cuda.is_available(): peak = torch.cuda.max_memory_allocated() / 1e6 else: peak = -1 after = nvidia_mem_used_mb() print(f" Pico VRAM (torch): {peak:.0f} MB") print(f" VRAM nvidia-smi: antes={before} MB → tras generar={after} MB (delta {after - before} MB)") def measure_gguf(path, n_gpu_layers=999, max_new=24, ckpt="/tmp/vram.ckpt"): binary = "/home/methodwhite/.local/bin/llama-cli" if not Path(binary).exists(): print(f"llama-cli no encontrado en {binary}") return cmd = [ binary, "-m", str(path), "-n", str(max_new), "-p", "What is the capital of France?", "-ngl", str(n_gpu_layers), "--no-display-prompt", "--no-warmup", "-c", "512", ] print(f"Ejecutando: {' '.join(cmd)}") before = nvidia_mem_used_mb() proc = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True) peak = before try: while True: line = proc.stdout.readline() if not line: break line = line.strip() if line and "log" not in line.lower() and not line.startswith("llama_"): pass used = nvidia_mem_used_mb() if used > peak: peak = used if proc.poll() is not None: break time.sleep(0.2) finally: proc.kill() print(f" VRAM pico (nvidia-smi): {peak} MB (antes {before} MB)") def main(): parser = argparse.ArgumentParser() parser.add_argument("--hf-dir", type=str) parser.add_argument("--nf4", action="store_true", help="Cargar en 4-bit NF4 en vez de bf16") parser.add_argument("--gguf", type=str) parser.add_argument("--n-gpu-layers", type=int, default=999) args = parser.parse_args() if args.hf_dir: measure_transformers(args.hf_dir, nf4=args.nf4) elif args.gguf: measure_gguf(args.gguf, args.n_gpu_layers) else: parser.print_help() if __name__ == "__main__": main()