import argparse import sys import time try: import torch except Exception as e: print(f"导入PyTorch失败: {e}") sys.exit(1) def bytes_to_gb(x: int) -> float: return x / 1024 ** 3 def clear_all_gpus(verbose: bool = True, sync: bool = True) -> None: if not torch.cuda.is_available(): print("CUDA不可用,无需清理。") return num_gpus = torch.cuda.device_count() if verbose: print(f"检测到 {num_gpus} 个GPU,开始清理缓存…") for i in range(num_gpus): try: torch.cuda.set_device(i) before_alloc = torch.cuda.memory_allocated(i) before_reserved = torch.cuda.memory_reserved(i) # 清理缓存与IPC句柄 torch.cuda.empty_cache() torch.cuda.ipc_collect() if sync: torch.cuda.synchronize(i) after_alloc = torch.cuda.memory_allocated(i) after_reserved = torch.cuda.memory_reserved(i) if verbose: print( f"GPU {i}: 已分配 {bytes_to_gb(before_alloc):.2f}GB -> {bytes_to_gb(after_alloc):.2f}GB, " f"已保留 {bytes_to_gb(before_reserved):.2f}GB -> {bytes_to_gb(after_reserved):.2f}GB" ) except Exception as e: print(f"清理GPU {i} 时出错: {e}") if verbose: print("清理完成。") def main() -> int: parser = argparse.ArgumentParser(description="清理所有GPU缓存与IPC资源") parser.add_argument("--quiet", action="store_true", help="安静模式,不输出详细信息") parser.add_argument("--no-sync", action="store_true", help="清理后不执行cuda同步") parser.add_argument("--repeat", type=int, default=1, help="重复清理次数(默认1次)") parser.add_argument( "--interval", type=float, default=0.0, help="重复清理时每次间隔秒数(默认0)" ) args = parser.parse_args() verbose = not args.quiet sync = not args.__dict__["no_sync"] if not torch.cuda.is_available(): print("CUDA不可用,无需清理。") return 0 for r in range(args.repeat): if args.repeat > 1 and verbose: print(f"第 {r + 1}/{args.repeat} 次清理…") clear_all_gpus(verbose=verbose, sync=sync) if r < args.repeat - 1 and args.interval > 0: time.sleep(args.interval) return 0 if __name__ == "__main__": sys.exit(main())