Download gpu_utils/clean_gpu.py from hulehule/pllm2-full-dump: direct link, hf CLI and curl.
- Browser
- Download file 2.5 kB
-
https://huggingface.co/hulehule/pllm2-full-dump/resolve/main/gpu_utils/clean_gpu.py
- Command line
-
hf download hf://hulehule/pllm2-full-dump/gpu_utils/clean_gpu.py
-
curl -L -o clean_gpu.py https://huggingface.co/hulehule/pllm2-full-dump/resolve/main/gpu_utils/clean_gpu.py
2.5 kB
| import argparse | |
| import sys | |
| import time | |
| try: | |
| import torch | |
| except Exception as e: | |
| print(f"导入PyTorch失败: {e}") | |
| sys.exit(1) | |
| def bytes_to_gb(x: int) -> float: | |
| return x / 1024 ** 3 | |
| def clear_all_gpus(verbose: bool = True, sync: bool = True) -> None: | |
| if not torch.cuda.is_available(): | |
| print("CUDA不可用,无需清理。") | |
| return | |
| num_gpus = torch.cuda.device_count() | |
| if verbose: | |
| print(f"检测到 {num_gpus} 个GPU,开始清理缓存…") | |
| for i in range(num_gpus): | |
| try: | |
| torch.cuda.set_device(i) | |
| before_alloc = torch.cuda.memory_allocated(i) | |
| before_reserved = torch.cuda.memory_reserved(i) | |
| # 清理缓存与IPC句柄 | |
| torch.cuda.empty_cache() | |
| torch.cuda.ipc_collect() | |
| if sync: | |
| torch.cuda.synchronize(i) | |
| after_alloc = torch.cuda.memory_allocated(i) | |
| after_reserved = torch.cuda.memory_reserved(i) | |
| if verbose: | |
| print( | |
| f"GPU {i}: 已分配 {bytes_to_gb(before_alloc):.2f}GB -> {bytes_to_gb(after_alloc):.2f}GB, " | |
| f"已保留 {bytes_to_gb(before_reserved):.2f}GB -> {bytes_to_gb(after_reserved):.2f}GB" | |
| ) | |
| except Exception as e: | |
| print(f"清理GPU {i} 时出错: {e}") | |
| if verbose: | |
| print("清理完成。") | |
| def main() -> int: | |
| parser = argparse.ArgumentParser(description="清理所有GPU缓存与IPC资源") | |
| parser.add_argument("--quiet", action="store_true", help="安静模式,不输出详细信息") | |
| parser.add_argument("--no-sync", action="store_true", help="清理后不执行cuda同步") | |
| parser.add_argument("--repeat", type=int, default=1, help="重复清理次数(默认1次)") | |
| parser.add_argument( | |
| "--interval", type=float, default=0.0, help="重复清理时每次间隔秒数(默认0)" | |
| ) | |
| args = parser.parse_args() | |
| verbose = not args.quiet | |
| sync = not args.__dict__["no_sync"] | |
| if not torch.cuda.is_available(): | |
| print("CUDA不可用,无需清理。") | |
| return 0 | |
| for r in range(args.repeat): | |
| if args.repeat > 1 and verbose: | |
| print(f"第 {r + 1}/{args.repeat} 次清理…") | |
| clear_all_gpus(verbose=verbose, sync=sync) | |
| if r < args.repeat - 1 and args.interval > 0: | |
| time.sleep(args.interval) | |
| return 0 | |
| if __name__ == "__main__": | |
| sys.exit(main()) | |