BiGRU_T_version / src /bigru_t /utils /hardware_detector.py
PowerMachine's picture
Upload folder using huggingface_hub
3275441 verified
Raw History Blame Contribute Delete
14.3 kB
"""
hardware_detector — Auto-detecção de hardware e otimização para PyTorch.
Detecta automaticamente:
- CPU (sempre disponível)
- CUDA (GPU NVIDIA)
- MPS (Apple Silicon)
- XPU (Intel)
- Número ótimo de threads
- Memória disponível
- Docker/Container detection (optXeon ex. 3, 4)
- Intel Xeon detection (optXeon)
- AVX-512/VNNI/AMX flags (optXeon ex. 2)
- IPEX availability (optXeon ex. 6)
Aplica otimizações:
- torch.set_num_threads() otimizado (cgroups v2 para Docker)
- OMP_PROC_BIND=CLOSE, OMP_PLACES=CORES (Xeon Docker)
- torch.backends.mkldnn.enabled = True (Intel oneDNN)
- torch.set_num_interop_threads(1) (crítico em Docker)
- torch.backends.cudnn.benchmark = True (se GPU)
- Mixed precision (torch.cuda.amp) se GPU
- CPU AMP (torch.cpu.amp.autocast) para Xeon (optXeon ex. 6, 7)
Uso:
from flexnet.hardware_detector import HardwareDetector, get_device, to_device
device = get_device() # torch.device('cuda' | 'mps' | 'cpu')
model = model.to(device)
batch = to_device(batch, device)
"""
from __future__ import annotations
import os
import sys
import gc
import math
import platform
import subprocess
from typing import Dict, Any, Optional, Tuple
import torch
def _get_real_cores_docker() -> int:
"""Detecta cores reais disponíveis em Docker (optXeon ex. 3).
Em containers Docker (cgroups v2), os.cpu_count() pode retornar
mais cores do que o container tem. Esta função resolve isso.
"""
cores = os.cpu_count() or 1
if not os.path.exists("/.dockerenv"):
return cores
# cpuset: lista de cores alocados
cpuset_path = "/sys/fs/cgroup/cpuset.cpus"
if os.path.exists(cpuset_path):
try:
with open(cpuset_path, "r") as f:
conteudo = f.read().strip()
if conteudo:
cores_lista = []
for parte in conteudo.split(","):
if "-" in parte:
inicio, fim = map(int, parte.split("-"))
cores_lista.extend(range(inicio, fim + 1))
else:
cores_lista.append(int(parte))
if cores_lista:
return len(cores_lista)
except Exception:
pass
# cgroups v2 quota/period
quota_path = "/sys/fs/cgroup/cpu.max"
if os.path.exists(quota_path):
try:
with open(quota_path, "r") as f:
valores = f.read().strip().split()
if len(valores) == 2 and valores[0] != "max":
quota = int(valores[0])
periodo = int(valores[1])
return max(1, math.ceil(quota / periodo))
except Exception:
pass
return cores
def _detect_xeon_cpu() -> Tuple[str, bool]:
"""Detecta modelo da CPU e se é Intel Xeon."""
nome_cpu = "Desconhecido"
try:
with open("/proc/cpuinfo", "r", encoding="utf-8") as f:
for line in f:
if "model name" in line.lower():
nome_cpu = line.split(":")[-1].strip()
break
except Exception:
nome_cpu = platform.processor()
is_xeon = "xeon" in nome_cpu.lower() or "intel" in nome_cpu.lower()
return nome_cpu, is_xeon
def _detect_avx512_vnni_amx() -> Tuple[bool, bool, bool]:
"""Detecta AVX-512, VNNI, AMX flags (optXeon ex. 2)."""
has_avx512 = False
has_vnni = False
has_amx = False
try:
with open("/proc/cpuinfo", "r", encoding="utf-8") as f:
for line in f:
if "flags" in line.lower():
flags = line.split()
has_avx512 = "avx512f" in flags
has_vnni = "avx512vnni" in flags
has_amx = "amx_int8" in flags
break
except Exception:
pass
return has_avx512, has_vnni, has_amx
def _detect_ipex() -> bool:
"""Detecta Intel Extension for PyTorch (optXeon ex. 6)."""
try:
import intel_extension_for_pytorch as ipex
return True
except ImportError:
return False
class HardwareDetector:
"""Auto-detecção de hardware e otimização para PyTorch.
Aprimorado (optXeon):
- Docker/Container detection
- Xeon CPU detection + thread config
- AVX-512/VNNI/AMX detection
- IPEX availability
- Xeon thread optimization (OMP_PROC_BIND=CLOSE, mkldnn)
"""
_instance: Optional['HardwareDetector'] = None
_device: Optional[torch.device] = None
_info: Optional[Dict] = None
@classmethod
def get_instance(cls) -> 'HardwareDetector':
if cls._instance is None:
cls._instance = cls()
return cls._instance
def __init__(self):
self._detect()
def _detect(self):
"""Detecta hardware disponível."""
# Docker/Xeon/AVX-512 detection (optXeon)
cpu_name, is_xeon = _detect_xeon_cpu()
has_avx512, has_vnni, has_amx = _detect_avx512_vnni_amx()
is_docker = os.path.exists("/.dockerenv")
cores_real = _get_real_cores_docker()
has_ipex = _detect_ipex()
self._info = {
'python_version': sys.version.split()[0],
'gil_enabled': sys._is_gil_enabled() if hasattr(sys, '_is_gil_enabled') else None,
'torch_version': torch.__version__,
'thp_mem_alloc': os.environ.get('THP_MEM_ALLOC_ENABLE', '0'),
'cpu_model_name': cpu_name,
'is_xeon': is_xeon,
'is_docker': is_docker,
'cores_real': cores_real,
'has_avx512': has_avx512,
'has_vnni': has_vnni,
'has_amx': has_amx,
'has_bf16_native': has_avx512 or is_xeon,
'has_ipex': has_ipex,
}
# Detectar device.
if torch.cuda.is_available():
self._device = torch.device('cuda')
self._info['device'] = 'cuda'
self._info['cuda_device_count'] = torch.cuda.device_count()
self._info['cuda_device_name'] = torch.cuda.get_device_name(0)
props = torch.cuda.get_device_properties(0)
self._info['cuda_memory_total_gb'] = round(props.total_memory / 1e9, 1)
self._info['cuda_compute_capability'] = f"{props.major}.{props.minor}"
torch.backends.cudnn.benchmark = True
torch.backends.cudnn.deterministic = False
elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available():
self._device = torch.device('mps')
self._info['device'] = 'mps'
else:
self._device = torch.device('cpu')
self._info['device'] = 'cpu'
# CPU info.
self._info['cpu_count'] = os.cpu_count()
self._info['torch_threads'] = torch.get_num_threads()
# Xeon thread optimization (optXeon ex. 3, 4)
if is_xeon and self._device.type == 'cpu':
threads_target = min(cores_real, 16) # max_threads=16
if threads_target > 16:
threads_target = 16
# OMP/MKL thread configuration
os.environ["OMP_NUM_THREADS"] = str(threads_target)
os.environ["MKL_NUM_THREADS"] = str(threads_target)
os.environ["OPENBLAS_NUM_THREADS"] = str(threads_target)
os.environ["VECLIB_MAXIMUM_THREADS"] = str(threads_target)
os.environ["NUMEXPR_NUM_THREADS"] = str(threads_target)
# Afinidade compacta (optXeon ex. 4)
if is_docker:
os.environ["OMP_PROC_BIND"] = "CLOSE"
os.environ["OMP_PLACES"] = "CORES"
torch.set_num_threads(threads_target)
torch.set_num_interop_threads(1) # Crítico no Docker (optXeon ex. 4)
torch.backends.mkldnn.enabled = True
self._info['xeon_threads_configured'] = threads_target
logger_msg = f"Xeon threads: {threads_target}, OMP_PROC_BIND={os.environ.get('OMP_PROC_BIND', 'N/A')}"
elif self._device.type == 'cpu':
# Non-Xeon CPU optimization
if not sys._is_gil_enabled() if hasattr(sys, '_is_gil_enabled') else False:
torch.set_num_threads(os.cpu_count())
else:
torch.set_num_threads(max(1, os.cpu_count() // 2))
self._info['torch_threads_optimized'] = torch.get_num_threads()
# Memória.
try:
import psutil
vm = psutil.virtual_memory()
self._info['ram_total_gb'] = round(vm.total / 1e9, 1)
self._info['ram_available_gb'] = round(vm.available / 1e9, 1)
except ImportError:
self._info['ram_total_gb'] = None
@property
def device(self) -> torch.device:
return self._device
@property
def info(self) -> Dict:
return self._info
@property
def is_gpu(self) -> bool:
return self._device.type in ('cuda', 'mps')
@property
def is_cuda(self) -> bool:
return self._device.type == 'cuda'
@property
def is_xeon(self) -> bool:
return self._info.get('is_xeon', False)
@property
def has_avx512(self) -> bool:
return self._info.get('has_avx512', False)
@property
def has_ipex(self) -> bool:
return self._info.get('has_ipex', False)
@property
def is_docker(self) -> bool:
return self._info.get('is_docker', False)
@property
def is_free_threaded(self) -> bool:
return hasattr(sys, '_is_gil_enabled') and not sys._is_gil_enabled()
def to_device(self, obj):
"""Move tensor, model, ou dict de tensors para o device detectado."""
if isinstance(obj, torch.Tensor):
return obj.to(self._device)
elif isinstance(obj, torch.nn.Module):
return obj.to(self._device)
elif isinstance(obj, dict):
return {k: self.to_device(v) for k, v in obj.items()}
elif isinstance(obj, (list, tuple)):
return type(obj)(self.to_device(v) for v in obj)
return obj
def cleanup(self):
"""Limpeza de memória otimizada para o device detectado."""
gc.collect(0)
gc.collect(1)
gc.collect(2)
if self.is_cuda:
torch.cuda.empty_cache()
torch.cuda.synchronize()
def get_mixed_precision_context(self):
"""Retorna context manager para mixed precision (se GPU ou Xeon CPU)."""
if self.is_cuda:
return torch.cuda.amp.autocast()
# Xeon CPU: BFloat16 via AVX-512 (optXeon ex. 6, 7)
if self.is_xeon and self.has_avx512:
try:
return torch.cpu.amp.autocast(dtype=torch.bfloat16)
except AttributeError:
pass
# CPU genérico: context nulo.
from contextlib import nullcontext
return nullcontext()
def optimize_for_inference(self, model: torch.nn.Module) -> torch.nn.Module:
"""Otimiza modelo para inferência.
Aprimorado (optXeon):
- IPEX optimization para Xeon
- torch.compile(mode="reduce-overhead") para AVX-512
- IPEX JIT trace para ultra-baixa latência
"""
model = model.to(self._device)
model.eval()
# IPEX optimization (optXeon ex. 6, 8)
if self.has_ipex and self._device.type == 'cpu':
try:
import intel_extension_for_pytorch as ipex
model = ipex.optimize(model, dtype=torch.bfloat16, inplace=True)
# JIT trace para inferência (optXeon ex. 8)
logger.info("IPEX: modelo otimizado para inferência (BFloat16 + prepacking)")
except Exception:
pass
# torch.compile se disponível (PyTorch 2.0+)
if hasattr(torch, 'compile'):
try:
compile_mode = "reduce-overhead" if self.has_avx512 else "default"
model = torch.compile(model, mode=compile_mode)
except Exception:
pass
return model
def get_stats(self) -> Dict:
"""Retorna stats de hardware em tempo real."""
stats = dict(self._info)
try:
import psutil
p = psutil.Process()
stats['rss_mb'] = round(p.memory_info().rss / 1e6, 1)
except ImportError:
stats['rss_mb'] = 0.0
if self.is_cuda:
stats['cuda_allocated_mb'] = round(torch.cuda.memory_allocated() / 1e6, 1)
stats['cuda_reserved_mb'] = round(torch.cuda.memory_reserved() / 1e6, 1)
return stats
# ============================================================
# Funções de conveniência
# ============================================================
def get_device() -> torch.device:
"""Retorna o device detectado (cuda | mps | cpu)."""
return HardwareDetector.get_instance().device
def to_device(obj):
"""Move tensor/model/dict para o device detectado."""
return HardwareDetector.get_instance().to_device(obj)
def cleanup():
"""Limpeza de memória otimizada."""
HardwareDetector.get_instance().cleanup()
def get_hardware_info() -> Dict:
"""Retorna info de hardware."""
return HardwareDetector.get_instance().info
def is_free_threaded() -> bool:
"""Verifica se Python está em modo free-threaded (GIL off)."""
return HardwareDetector.get_instance().is_free_threaded
def is_gpu() -> bool:
"""Verifica se GPU está disponível."""
return HardwareDetector.get_instance().is_gpu
def is_xeon() -> bool:
"""Verifica se CPU é Intel Xeon (optXeon)."""
return HardwareDetector.get_instance().is_xeon
def has_avx512() -> bool:
"""Verifica se CPU suporta AVX-512 (optXeon)."""
return HardwareDetector.get_instance().has_avx512
def has_ipex() -> bool:
"""Verifica se IPEX está disponível (optXeon)."""
return HardwareDetector.get_instance().has_ipex
def is_docker() -> bool:
"""Verifica se está rodando em Docker (optXeon)."""
return HardwareDetector.get_instance().is_docker
def optimize_for_inference(model: torch.nn.Module) -> torch.nn.Module:
"""Otimiza modelo para inferência no device detectado."""
return HardwareDetector.get_instance().optimize_for_inference(model)