""" hardware_detector — Auto-detecção de hardware e otimização para PyTorch. Detecta automaticamente: - CPU (sempre disponível) - CUDA (GPU NVIDIA) - MPS (Apple Silicon) - XPU (Intel) - Número ótimo de threads - Memória disponível - Docker/Container detection (optXeon ex. 3, 4) - Intel Xeon detection (optXeon) - AVX-512/VNNI/AMX flags (optXeon ex. 2) - IPEX availability (optXeon ex. 6) Aplica otimizações: - torch.set_num_threads() otimizado (cgroups v2 para Docker) - OMP_PROC_BIND=CLOSE, OMP_PLACES=CORES (Xeon Docker) - torch.backends.mkldnn.enabled = True (Intel oneDNN) - torch.set_num_interop_threads(1) (crítico em Docker) - torch.backends.cudnn.benchmark = True (se GPU) - Mixed precision (torch.cuda.amp) se GPU - CPU AMP (torch.cpu.amp.autocast) para Xeon (optXeon ex. 6, 7) Uso: from flexnet.hardware_detector import HardwareDetector, get_device, to_device device = get_device() # torch.device('cuda' | 'mps' | 'cpu') model = model.to(device) batch = to_device(batch, device) """ from __future__ import annotations import os import sys import gc import math import platform import subprocess from typing import Dict, Any, Optional, Tuple import torch def _get_real_cores_docker() -> int: """Detecta cores reais disponíveis em Docker (optXeon ex. 3). Em containers Docker (cgroups v2), os.cpu_count() pode retornar mais cores do que o container tem. Esta função resolve isso. """ cores = os.cpu_count() or 1 if not os.path.exists("/.dockerenv"): return cores # cpuset: lista de cores alocados cpuset_path = "/sys/fs/cgroup/cpuset.cpus" if os.path.exists(cpuset_path): try: with open(cpuset_path, "r") as f: conteudo = f.read().strip() if conteudo: cores_lista = [] for parte in conteudo.split(","): if "-" in parte: inicio, fim = map(int, parte.split("-")) cores_lista.extend(range(inicio, fim + 1)) else: cores_lista.append(int(parte)) if cores_lista: return len(cores_lista) except Exception: pass # cgroups v2 quota/period quota_path = "/sys/fs/cgroup/cpu.max" if os.path.exists(quota_path): try: with open(quota_path, "r") as f: valores = f.read().strip().split() if len(valores) == 2 and valores[0] != "max": quota = int(valores[0]) periodo = int(valores[1]) return max(1, math.ceil(quota / periodo)) except Exception: pass return cores def _detect_xeon_cpu() -> Tuple[str, bool]: """Detecta modelo da CPU e se é Intel Xeon.""" nome_cpu = "Desconhecido" try: with open("/proc/cpuinfo", "r", encoding="utf-8") as f: for line in f: if "model name" in line.lower(): nome_cpu = line.split(":")[-1].strip() break except Exception: nome_cpu = platform.processor() is_xeon = "xeon" in nome_cpu.lower() or "intel" in nome_cpu.lower() return nome_cpu, is_xeon def _detect_avx512_vnni_amx() -> Tuple[bool, bool, bool]: """Detecta AVX-512, VNNI, AMX flags (optXeon ex. 2).""" has_avx512 = False has_vnni = False has_amx = False try: with open("/proc/cpuinfo", "r", encoding="utf-8") as f: for line in f: if "flags" in line.lower(): flags = line.split() has_avx512 = "avx512f" in flags has_vnni = "avx512vnni" in flags has_amx = "amx_int8" in flags break except Exception: pass return has_avx512, has_vnni, has_amx def _detect_ipex() -> bool: """Detecta Intel Extension for PyTorch (optXeon ex. 6).""" try: import intel_extension_for_pytorch as ipex return True except ImportError: return False class HardwareDetector: """Auto-detecção de hardware e otimização para PyTorch. Aprimorado (optXeon): - Docker/Container detection - Xeon CPU detection + thread config - AVX-512/VNNI/AMX detection - IPEX availability - Xeon thread optimization (OMP_PROC_BIND=CLOSE, mkldnn) """ _instance: Optional['HardwareDetector'] = None _device: Optional[torch.device] = None _info: Optional[Dict] = None @classmethod def get_instance(cls) -> 'HardwareDetector': if cls._instance is None: cls._instance = cls() return cls._instance def __init__(self): self._detect() def _detect(self): """Detecta hardware disponível.""" # Docker/Xeon/AVX-512 detection (optXeon) cpu_name, is_xeon = _detect_xeon_cpu() has_avx512, has_vnni, has_amx = _detect_avx512_vnni_amx() is_docker = os.path.exists("/.dockerenv") cores_real = _get_real_cores_docker() has_ipex = _detect_ipex() self._info = { 'python_version': sys.version.split()[0], 'gil_enabled': sys._is_gil_enabled() if hasattr(sys, '_is_gil_enabled') else None, 'torch_version': torch.__version__, 'thp_mem_alloc': os.environ.get('THP_MEM_ALLOC_ENABLE', '0'), 'cpu_model_name': cpu_name, 'is_xeon': is_xeon, 'is_docker': is_docker, 'cores_real': cores_real, 'has_avx512': has_avx512, 'has_vnni': has_vnni, 'has_amx': has_amx, 'has_bf16_native': has_avx512 or is_xeon, 'has_ipex': has_ipex, } # Detectar device. if torch.cuda.is_available(): self._device = torch.device('cuda') self._info['device'] = 'cuda' self._info['cuda_device_count'] = torch.cuda.device_count() self._info['cuda_device_name'] = torch.cuda.get_device_name(0) props = torch.cuda.get_device_properties(0) self._info['cuda_memory_total_gb'] = round(props.total_memory / 1e9, 1) self._info['cuda_compute_capability'] = f"{props.major}.{props.minor}" torch.backends.cudnn.benchmark = True torch.backends.cudnn.deterministic = False elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available(): self._device = torch.device('mps') self._info['device'] = 'mps' else: self._device = torch.device('cpu') self._info['device'] = 'cpu' # CPU info. self._info['cpu_count'] = os.cpu_count() self._info['torch_threads'] = torch.get_num_threads() # Xeon thread optimization (optXeon ex. 3, 4) if is_xeon and self._device.type == 'cpu': threads_target = min(cores_real, 16) # max_threads=16 if threads_target > 16: threads_target = 16 # OMP/MKL thread configuration os.environ["OMP_NUM_THREADS"] = str(threads_target) os.environ["MKL_NUM_THREADS"] = str(threads_target) os.environ["OPENBLAS_NUM_THREADS"] = str(threads_target) os.environ["VECLIB_MAXIMUM_THREADS"] = str(threads_target) os.environ["NUMEXPR_NUM_THREADS"] = str(threads_target) # Afinidade compacta (optXeon ex. 4) if is_docker: os.environ["OMP_PROC_BIND"] = "CLOSE" os.environ["OMP_PLACES"] = "CORES" torch.set_num_threads(threads_target) torch.set_num_interop_threads(1) # Crítico no Docker (optXeon ex. 4) torch.backends.mkldnn.enabled = True self._info['xeon_threads_configured'] = threads_target logger_msg = f"Xeon threads: {threads_target}, OMP_PROC_BIND={os.environ.get('OMP_PROC_BIND', 'N/A')}" elif self._device.type == 'cpu': # Non-Xeon CPU optimization if not sys._is_gil_enabled() if hasattr(sys, '_is_gil_enabled') else False: torch.set_num_threads(os.cpu_count()) else: torch.set_num_threads(max(1, os.cpu_count() // 2)) self._info['torch_threads_optimized'] = torch.get_num_threads() # Memória. try: import psutil vm = psutil.virtual_memory() self._info['ram_total_gb'] = round(vm.total / 1e9, 1) self._info['ram_available_gb'] = round(vm.available / 1e9, 1) except ImportError: self._info['ram_total_gb'] = None @property def device(self) -> torch.device: return self._device @property def info(self) -> Dict: return self._info @property def is_gpu(self) -> bool: return self._device.type in ('cuda', 'mps') @property def is_cuda(self) -> bool: return self._device.type == 'cuda' @property def is_xeon(self) -> bool: return self._info.get('is_xeon', False) @property def has_avx512(self) -> bool: return self._info.get('has_avx512', False) @property def has_ipex(self) -> bool: return self._info.get('has_ipex', False) @property def is_docker(self) -> bool: return self._info.get('is_docker', False) @property def is_free_threaded(self) -> bool: return hasattr(sys, '_is_gil_enabled') and not sys._is_gil_enabled() def to_device(self, obj): """Move tensor, model, ou dict de tensors para o device detectado.""" if isinstance(obj, torch.Tensor): return obj.to(self._device) elif isinstance(obj, torch.nn.Module): return obj.to(self._device) elif isinstance(obj, dict): return {k: self.to_device(v) for k, v in obj.items()} elif isinstance(obj, (list, tuple)): return type(obj)(self.to_device(v) for v in obj) return obj def cleanup(self): """Limpeza de memória otimizada para o device detectado.""" gc.collect(0) gc.collect(1) gc.collect(2) if self.is_cuda: torch.cuda.empty_cache() torch.cuda.synchronize() def get_mixed_precision_context(self): """Retorna context manager para mixed precision (se GPU ou Xeon CPU).""" if self.is_cuda: return torch.cuda.amp.autocast() # Xeon CPU: BFloat16 via AVX-512 (optXeon ex. 6, 7) if self.is_xeon and self.has_avx512: try: return torch.cpu.amp.autocast(dtype=torch.bfloat16) except AttributeError: pass # CPU genérico: context nulo. from contextlib import nullcontext return nullcontext() def optimize_for_inference(self, model: torch.nn.Module) -> torch.nn.Module: """Otimiza modelo para inferência. Aprimorado (optXeon): - IPEX optimization para Xeon - torch.compile(mode="reduce-overhead") para AVX-512 - IPEX JIT trace para ultra-baixa latência """ model = model.to(self._device) model.eval() # IPEX optimization (optXeon ex. 6, 8) if self.has_ipex and self._device.type == 'cpu': try: import intel_extension_for_pytorch as ipex model = ipex.optimize(model, dtype=torch.bfloat16, inplace=True) # JIT trace para inferência (optXeon ex. 8) logger.info("IPEX: modelo otimizado para inferência (BFloat16 + prepacking)") except Exception: pass # torch.compile se disponível (PyTorch 2.0+) if hasattr(torch, 'compile'): try: compile_mode = "reduce-overhead" if self.has_avx512 else "default" model = torch.compile(model, mode=compile_mode) except Exception: pass return model def get_stats(self) -> Dict: """Retorna stats de hardware em tempo real.""" stats = dict(self._info) try: import psutil p = psutil.Process() stats['rss_mb'] = round(p.memory_info().rss / 1e6, 1) except ImportError: stats['rss_mb'] = 0.0 if self.is_cuda: stats['cuda_allocated_mb'] = round(torch.cuda.memory_allocated() / 1e6, 1) stats['cuda_reserved_mb'] = round(torch.cuda.memory_reserved() / 1e6, 1) return stats # ============================================================ # Funções de conveniência # ============================================================ def get_device() -> torch.device: """Retorna o device detectado (cuda | mps | cpu).""" return HardwareDetector.get_instance().device def to_device(obj): """Move tensor/model/dict para o device detectado.""" return HardwareDetector.get_instance().to_device(obj) def cleanup(): """Limpeza de memória otimizada.""" HardwareDetector.get_instance().cleanup() def get_hardware_info() -> Dict: """Retorna info de hardware.""" return HardwareDetector.get_instance().info def is_free_threaded() -> bool: """Verifica se Python está em modo free-threaded (GIL off).""" return HardwareDetector.get_instance().is_free_threaded def is_gpu() -> bool: """Verifica se GPU está disponível.""" return HardwareDetector.get_instance().is_gpu def is_xeon() -> bool: """Verifica se CPU é Intel Xeon (optXeon).""" return HardwareDetector.get_instance().is_xeon def has_avx512() -> bool: """Verifica se CPU suporta AVX-512 (optXeon).""" return HardwareDetector.get_instance().has_avx512 def has_ipex() -> bool: """Verifica se IPEX está disponível (optXeon).""" return HardwareDetector.get_instance().has_ipex def is_docker() -> bool: """Verifica se está rodando em Docker (optXeon).""" return HardwareDetector.get_instance().is_docker def optimize_for_inference(model: torch.nn.Module) -> torch.nn.Module: """Otimiza modelo para inferência no device detectado.""" return HardwareDetector.get_instance().optimize_for_inference(model)