Download src/bigru_t/utils/hardware_detector.py from PowerMachine/BiGRU_T_version: direct link, hf CLI and curl.
- Browser
- Download file 14.3 kB
-
https://huggingface.co/PowerMachine/BiGRU_T_version/resolve/main/src/bigru_t/utils/hardware_detector.py
- Command line
-
hf download hf://PowerMachine/BiGRU_T_version/src/bigru_t/utils/hardware_detector.py
-
curl -L -o hardware_detector.py https://huggingface.co/PowerMachine/BiGRU_T_version/resolve/main/src/bigru_t/utils/hardware_detector.py
14.3 kB
| """ | |
| hardware_detector — Auto-detecção de hardware e otimização para PyTorch. | |
| Detecta automaticamente: | |
| - CPU (sempre disponível) | |
| - CUDA (GPU NVIDIA) | |
| - MPS (Apple Silicon) | |
| - XPU (Intel) | |
| - Número ótimo de threads | |
| - Memória disponível | |
| - Docker/Container detection (optXeon ex. 3, 4) | |
| - Intel Xeon detection (optXeon) | |
| - AVX-512/VNNI/AMX flags (optXeon ex. 2) | |
| - IPEX availability (optXeon ex. 6) | |
| Aplica otimizações: | |
| - torch.set_num_threads() otimizado (cgroups v2 para Docker) | |
| - OMP_PROC_BIND=CLOSE, OMP_PLACES=CORES (Xeon Docker) | |
| - torch.backends.mkldnn.enabled = True (Intel oneDNN) | |
| - torch.set_num_interop_threads(1) (crítico em Docker) | |
| - torch.backends.cudnn.benchmark = True (se GPU) | |
| - Mixed precision (torch.cuda.amp) se GPU | |
| - CPU AMP (torch.cpu.amp.autocast) para Xeon (optXeon ex. 6, 7) | |
| Uso: | |
| from flexnet.hardware_detector import HardwareDetector, get_device, to_device | |
| device = get_device() # torch.device('cuda' | 'mps' | 'cpu') | |
| model = model.to(device) | |
| batch = to_device(batch, device) | |
| """ | |
| from __future__ import annotations | |
| import os | |
| import sys | |
| import gc | |
| import math | |
| import platform | |
| import subprocess | |
| from typing import Dict, Any, Optional, Tuple | |
| import torch | |
| def _get_real_cores_docker() -> int: | |
| """Detecta cores reais disponíveis em Docker (optXeon ex. 3). | |
| Em containers Docker (cgroups v2), os.cpu_count() pode retornar | |
| mais cores do que o container tem. Esta função resolve isso. | |
| """ | |
| cores = os.cpu_count() or 1 | |
| if not os.path.exists("/.dockerenv"): | |
| return cores | |
| # cpuset: lista de cores alocados | |
| cpuset_path = "/sys/fs/cgroup/cpuset.cpus" | |
| if os.path.exists(cpuset_path): | |
| try: | |
| with open(cpuset_path, "r") as f: | |
| conteudo = f.read().strip() | |
| if conteudo: | |
| cores_lista = [] | |
| for parte in conteudo.split(","): | |
| if "-" in parte: | |
| inicio, fim = map(int, parte.split("-")) | |
| cores_lista.extend(range(inicio, fim + 1)) | |
| else: | |
| cores_lista.append(int(parte)) | |
| if cores_lista: | |
| return len(cores_lista) | |
| except Exception: | |
| pass | |
| # cgroups v2 quota/period | |
| quota_path = "/sys/fs/cgroup/cpu.max" | |
| if os.path.exists(quota_path): | |
| try: | |
| with open(quota_path, "r") as f: | |
| valores = f.read().strip().split() | |
| if len(valores) == 2 and valores[0] != "max": | |
| quota = int(valores[0]) | |
| periodo = int(valores[1]) | |
| return max(1, math.ceil(quota / periodo)) | |
| except Exception: | |
| pass | |
| return cores | |
| def _detect_xeon_cpu() -> Tuple[str, bool]: | |
| """Detecta modelo da CPU e se é Intel Xeon.""" | |
| nome_cpu = "Desconhecido" | |
| try: | |
| with open("/proc/cpuinfo", "r", encoding="utf-8") as f: | |
| for line in f: | |
| if "model name" in line.lower(): | |
| nome_cpu = line.split(":")[-1].strip() | |
| break | |
| except Exception: | |
| nome_cpu = platform.processor() | |
| is_xeon = "xeon" in nome_cpu.lower() or "intel" in nome_cpu.lower() | |
| return nome_cpu, is_xeon | |
| def _detect_avx512_vnni_amx() -> Tuple[bool, bool, bool]: | |
| """Detecta AVX-512, VNNI, AMX flags (optXeon ex. 2).""" | |
| has_avx512 = False | |
| has_vnni = False | |
| has_amx = False | |
| try: | |
| with open("/proc/cpuinfo", "r", encoding="utf-8") as f: | |
| for line in f: | |
| if "flags" in line.lower(): | |
| flags = line.split() | |
| has_avx512 = "avx512f" in flags | |
| has_vnni = "avx512vnni" in flags | |
| has_amx = "amx_int8" in flags | |
| break | |
| except Exception: | |
| pass | |
| return has_avx512, has_vnni, has_amx | |
| def _detect_ipex() -> bool: | |
| """Detecta Intel Extension for PyTorch (optXeon ex. 6).""" | |
| try: | |
| import intel_extension_for_pytorch as ipex | |
| return True | |
| except ImportError: | |
| return False | |
| class HardwareDetector: | |
| """Auto-detecção de hardware e otimização para PyTorch. | |
| Aprimorado (optXeon): | |
| - Docker/Container detection | |
| - Xeon CPU detection + thread config | |
| - AVX-512/VNNI/AMX detection | |
| - IPEX availability | |
| - Xeon thread optimization (OMP_PROC_BIND=CLOSE, mkldnn) | |
| """ | |
| _instance: Optional['HardwareDetector'] = None | |
| _device: Optional[torch.device] = None | |
| _info: Optional[Dict] = None | |
| def get_instance(cls) -> 'HardwareDetector': | |
| if cls._instance is None: | |
| cls._instance = cls() | |
| return cls._instance | |
| def __init__(self): | |
| self._detect() | |
| def _detect(self): | |
| """Detecta hardware disponível.""" | |
| # Docker/Xeon/AVX-512 detection (optXeon) | |
| cpu_name, is_xeon = _detect_xeon_cpu() | |
| has_avx512, has_vnni, has_amx = _detect_avx512_vnni_amx() | |
| is_docker = os.path.exists("/.dockerenv") | |
| cores_real = _get_real_cores_docker() | |
| has_ipex = _detect_ipex() | |
| self._info = { | |
| 'python_version': sys.version.split()[0], | |
| 'gil_enabled': sys._is_gil_enabled() if hasattr(sys, '_is_gil_enabled') else None, | |
| 'torch_version': torch.__version__, | |
| 'thp_mem_alloc': os.environ.get('THP_MEM_ALLOC_ENABLE', '0'), | |
| 'cpu_model_name': cpu_name, | |
| 'is_xeon': is_xeon, | |
| 'is_docker': is_docker, | |
| 'cores_real': cores_real, | |
| 'has_avx512': has_avx512, | |
| 'has_vnni': has_vnni, | |
| 'has_amx': has_amx, | |
| 'has_bf16_native': has_avx512 or is_xeon, | |
| 'has_ipex': has_ipex, | |
| } | |
| # Detectar device. | |
| if torch.cuda.is_available(): | |
| self._device = torch.device('cuda') | |
| self._info['device'] = 'cuda' | |
| self._info['cuda_device_count'] = torch.cuda.device_count() | |
| self._info['cuda_device_name'] = torch.cuda.get_device_name(0) | |
| props = torch.cuda.get_device_properties(0) | |
| self._info['cuda_memory_total_gb'] = round(props.total_memory / 1e9, 1) | |
| self._info['cuda_compute_capability'] = f"{props.major}.{props.minor}" | |
| torch.backends.cudnn.benchmark = True | |
| torch.backends.cudnn.deterministic = False | |
| elif hasattr(torch.backends, 'mps') and torch.backends.mps.is_available(): | |
| self._device = torch.device('mps') | |
| self._info['device'] = 'mps' | |
| else: | |
| self._device = torch.device('cpu') | |
| self._info['device'] = 'cpu' | |
| # CPU info. | |
| self._info['cpu_count'] = os.cpu_count() | |
| self._info['torch_threads'] = torch.get_num_threads() | |
| # Xeon thread optimization (optXeon ex. 3, 4) | |
| if is_xeon and self._device.type == 'cpu': | |
| threads_target = min(cores_real, 16) # max_threads=16 | |
| if threads_target > 16: | |
| threads_target = 16 | |
| # OMP/MKL thread configuration | |
| os.environ["OMP_NUM_THREADS"] = str(threads_target) | |
| os.environ["MKL_NUM_THREADS"] = str(threads_target) | |
| os.environ["OPENBLAS_NUM_THREADS"] = str(threads_target) | |
| os.environ["VECLIB_MAXIMUM_THREADS"] = str(threads_target) | |
| os.environ["NUMEXPR_NUM_THREADS"] = str(threads_target) | |
| # Afinidade compacta (optXeon ex. 4) | |
| if is_docker: | |
| os.environ["OMP_PROC_BIND"] = "CLOSE" | |
| os.environ["OMP_PLACES"] = "CORES" | |
| torch.set_num_threads(threads_target) | |
| torch.set_num_interop_threads(1) # Crítico no Docker (optXeon ex. 4) | |
| torch.backends.mkldnn.enabled = True | |
| self._info['xeon_threads_configured'] = threads_target | |
| logger_msg = f"Xeon threads: {threads_target}, OMP_PROC_BIND={os.environ.get('OMP_PROC_BIND', 'N/A')}" | |
| elif self._device.type == 'cpu': | |
| # Non-Xeon CPU optimization | |
| if not sys._is_gil_enabled() if hasattr(sys, '_is_gil_enabled') else False: | |
| torch.set_num_threads(os.cpu_count()) | |
| else: | |
| torch.set_num_threads(max(1, os.cpu_count() // 2)) | |
| self._info['torch_threads_optimized'] = torch.get_num_threads() | |
| # Memória. | |
| try: | |
| import psutil | |
| vm = psutil.virtual_memory() | |
| self._info['ram_total_gb'] = round(vm.total / 1e9, 1) | |
| self._info['ram_available_gb'] = round(vm.available / 1e9, 1) | |
| except ImportError: | |
| self._info['ram_total_gb'] = None | |
| def device(self) -> torch.device: | |
| return self._device | |
| def info(self) -> Dict: | |
| return self._info | |
| def is_gpu(self) -> bool: | |
| return self._device.type in ('cuda', 'mps') | |
| def is_cuda(self) -> bool: | |
| return self._device.type == 'cuda' | |
| def is_xeon(self) -> bool: | |
| return self._info.get('is_xeon', False) | |
| def has_avx512(self) -> bool: | |
| return self._info.get('has_avx512', False) | |
| def has_ipex(self) -> bool: | |
| return self._info.get('has_ipex', False) | |
| def is_docker(self) -> bool: | |
| return self._info.get('is_docker', False) | |
| def is_free_threaded(self) -> bool: | |
| return hasattr(sys, '_is_gil_enabled') and not sys._is_gil_enabled() | |
| def to_device(self, obj): | |
| """Move tensor, model, ou dict de tensors para o device detectado.""" | |
| if isinstance(obj, torch.Tensor): | |
| return obj.to(self._device) | |
| elif isinstance(obj, torch.nn.Module): | |
| return obj.to(self._device) | |
| elif isinstance(obj, dict): | |
| return {k: self.to_device(v) for k, v in obj.items()} | |
| elif isinstance(obj, (list, tuple)): | |
| return type(obj)(self.to_device(v) for v in obj) | |
| return obj | |
| def cleanup(self): | |
| """Limpeza de memória otimizada para o device detectado.""" | |
| gc.collect(0) | |
| gc.collect(1) | |
| gc.collect(2) | |
| if self.is_cuda: | |
| torch.cuda.empty_cache() | |
| torch.cuda.synchronize() | |
| def get_mixed_precision_context(self): | |
| """Retorna context manager para mixed precision (se GPU ou Xeon CPU).""" | |
| if self.is_cuda: | |
| return torch.cuda.amp.autocast() | |
| # Xeon CPU: BFloat16 via AVX-512 (optXeon ex. 6, 7) | |
| if self.is_xeon and self.has_avx512: | |
| try: | |
| return torch.cpu.amp.autocast(dtype=torch.bfloat16) | |
| except AttributeError: | |
| pass | |
| # CPU genérico: context nulo. | |
| from contextlib import nullcontext | |
| return nullcontext() | |
| def optimize_for_inference(self, model: torch.nn.Module) -> torch.nn.Module: | |
| """Otimiza modelo para inferência. | |
| Aprimorado (optXeon): | |
| - IPEX optimization para Xeon | |
| - torch.compile(mode="reduce-overhead") para AVX-512 | |
| - IPEX JIT trace para ultra-baixa latência | |
| """ | |
| model = model.to(self._device) | |
| model.eval() | |
| # IPEX optimization (optXeon ex. 6, 8) | |
| if self.has_ipex and self._device.type == 'cpu': | |
| try: | |
| import intel_extension_for_pytorch as ipex | |
| model = ipex.optimize(model, dtype=torch.bfloat16, inplace=True) | |
| # JIT trace para inferência (optXeon ex. 8) | |
| logger.info("IPEX: modelo otimizado para inferência (BFloat16 + prepacking)") | |
| except Exception: | |
| pass | |
| # torch.compile se disponível (PyTorch 2.0+) | |
| if hasattr(torch, 'compile'): | |
| try: | |
| compile_mode = "reduce-overhead" if self.has_avx512 else "default" | |
| model = torch.compile(model, mode=compile_mode) | |
| except Exception: | |
| pass | |
| return model | |
| def get_stats(self) -> Dict: | |
| """Retorna stats de hardware em tempo real.""" | |
| stats = dict(self._info) | |
| try: | |
| import psutil | |
| p = psutil.Process() | |
| stats['rss_mb'] = round(p.memory_info().rss / 1e6, 1) | |
| except ImportError: | |
| stats['rss_mb'] = 0.0 | |
| if self.is_cuda: | |
| stats['cuda_allocated_mb'] = round(torch.cuda.memory_allocated() / 1e6, 1) | |
| stats['cuda_reserved_mb'] = round(torch.cuda.memory_reserved() / 1e6, 1) | |
| return stats | |
| # ============================================================ | |
| # Funções de conveniência | |
| # ============================================================ | |
| def get_device() -> torch.device: | |
| """Retorna o device detectado (cuda | mps | cpu).""" | |
| return HardwareDetector.get_instance().device | |
| def to_device(obj): | |
| """Move tensor/model/dict para o device detectado.""" | |
| return HardwareDetector.get_instance().to_device(obj) | |
| def cleanup(): | |
| """Limpeza de memória otimizada.""" | |
| HardwareDetector.get_instance().cleanup() | |
| def get_hardware_info() -> Dict: | |
| """Retorna info de hardware.""" | |
| return HardwareDetector.get_instance().info | |
| def is_free_threaded() -> bool: | |
| """Verifica se Python está em modo free-threaded (GIL off).""" | |
| return HardwareDetector.get_instance().is_free_threaded | |
| def is_gpu() -> bool: | |
| """Verifica se GPU está disponível.""" | |
| return HardwareDetector.get_instance().is_gpu | |
| def is_xeon() -> bool: | |
| """Verifica se CPU é Intel Xeon (optXeon).""" | |
| return HardwareDetector.get_instance().is_xeon | |
| def has_avx512() -> bool: | |
| """Verifica se CPU suporta AVX-512 (optXeon).""" | |
| return HardwareDetector.get_instance().has_avx512 | |
| def has_ipex() -> bool: | |
| """Verifica se IPEX está disponível (optXeon).""" | |
| return HardwareDetector.get_instance().has_ipex | |
| def is_docker() -> bool: | |
| """Verifica se está rodando em Docker (optXeon).""" | |
| return HardwareDetector.get_instance().is_docker | |
| def optimize_for_inference(model: torch.nn.Module) -> torch.nn.Module: | |
| """Otimiza modelo para inferência no device detectado.""" | |
| return HardwareDetector.get_instance().optimize_for_inference(model) | |