Buckets:
| import time | |
| import gc | |
| import torch | |
| import numpy as np | |
| import pandas as pd | |
| from typing import Dict, List, Any, Callable, Optional, Tuple | |
| from collections import defaultdict | |
| import logging | |
| import psutil | |
| import os | |
| logger = logging.getLogger(__name__) | |
| class PerformanceBenchmark: | |
| def __init__(self): | |
| self.results = defaultdict(list) | |
| self.memory_snapshots = [] | |
| def benchmark_function( | |
| self, | |
| func: Callable, | |
| args: tuple = (), | |
| kwargs: dict = {}, | |
| name: str = "unnamed", | |
| warmup_runs: int = 3, | |
| benchmark_runs: int = 10, | |
| ) -> Tuple[Dict[str, Any], Any]: | |
| logger.info(f"Benchmarking {name}") | |
| for _ in range(warmup_runs): | |
| try: | |
| _ = func(*args, **kwargs) | |
| except Exception as e: | |
| logger.error(f"Warmup failed for {name}: {e}") | |
| return {}, None | |
| gc.collect() | |
| if torch.cuda.is_available(): | |
| torch.cuda.empty_cache() | |
| torch.cuda.synchronize() | |
| initial_memory = self._get_memory_usage() | |
| times = [] | |
| for run in range(benchmark_runs): | |
| start_time = time.perf_counter() | |
| try: | |
| result = func(*args, **kwargs) | |
| if torch.cuda.is_available(): | |
| torch.cuda.synchronize() | |
| end_time = time.perf_counter() | |
| times.append(end_time - start_time) | |
| except Exception as e: | |
| logger.error(f"Benchmark run {run} failed for {name}: {e}") | |
| continue | |
| final_memory = self._get_memory_usage() | |
| if not times: | |
| logger.error(f"No successful runs for {name}") | |
| return {}, None | |
| peak_memory = self._get_peak_memory() | |
| benchmark_result = { | |
| "name": name, | |
| "mean_time": np.mean(times), | |
| "std_time": np.std(times), | |
| "min_time": np.min(times), | |
| "max_time": np.max(times), | |
| "peak_cpu_memory": peak_memory["cpu"], | |
| "peak_gpu_memory": peak_memory["gpu"], | |
| "peak_process_memory": peak_memory["process"], | |
| "memory_delta": final_memory["process"] - initial_memory["process"], | |
| } | |
| self.results[name].append(benchmark_result) | |
| return benchmark_result, result | |
| def _get_memory_usage(self) -> Dict[str, float]: | |
| process = psutil.Process() | |
| memory_info = { | |
| "cpu": psutil.virtual_memory().used / (1024**2), | |
| "process": process.memory_info().rss / (1024**2), | |
| "gpu": 0.0, | |
| } | |
| if torch.cuda.is_available(): | |
| memory_info["gpu"] = torch.cuda.memory_allocated() / (1024**2) | |
| return memory_info | |
| def _get_peak_memory(self) -> Dict[str, float]: | |
| return self._get_memory_usage() | |
| def compare_results(self) -> Optional[pd.DataFrame]: | |
| if not self.results: | |
| logger.warning("No benchmark results available") | |
| return None | |
| df_data = [] | |
| for name, results in self.results.items(): | |
| for result in results: | |
| df_data.append(result) | |
| if not df_data: | |
| logger.warning("No benchmark data to compare") | |
| return None | |
| df = pd.DataFrame(df_data) | |
| logger.info("Benchmark Comparison Results:") | |
| logger.info("=" * 80) | |
| logger.info( | |
| f"{'Function':<20} {'Time (ms)':<12} {'Std (ms)':<10} {'Peak GPU (MB)':<15} {'Peak CPU (MB)':<15}" | |
| ) | |
| logger.info("-" * 80) | |
| for _, row in df.iterrows(): | |
| logger.info( | |
| f"{row['name']:<20} {row['mean_time']*1000:<12.3f} {row['std_time']*1000:<10.3f} " | |
| f"{row['peak_gpu_memory']:<15.1f} {row['peak_cpu_memory']:<15.1f}" | |
| ) | |
| return df | |
| def clear_results(self): | |
| self.results.clear() | |
| self.memory_snapshots.clear() | |
| class MemoryAccessAnalyzer: | |
| def analyze_spatial_locality(matrix_size: int = 2048) -> Dict[str, Any]: | |
| logger.info( | |
| f"Analyzing Spatial Locality (Matrix size: {matrix_size}x{matrix_size})" | |
| ) | |
| A = np.random.randn(matrix_size, matrix_size).astype(np.float32) | |
| def row_major_sum(matrix): | |
| total = 0.0 | |
| for i in range(matrix.shape[0]): | |
| for j in range(matrix.shape[1]): | |
| total += matrix[i, j] | |
| return total | |
| def column_major_sum(matrix): | |
| total = 0.0 | |
| for j in range(matrix.shape[1]): | |
| for i in range(matrix.shape[0]): | |
| total += matrix[i, j] | |
| return total | |
| def random_access_sum(matrix): | |
| indices = np.random.permutation(matrix_size * matrix_size) | |
| total = 0.0 | |
| for idx in indices: | |
| i, j = divmod(idx, matrix_size) | |
| total += matrix[i, j] | |
| return total | |
| benchmark = PerformanceBenchmark() | |
| results = {} | |
| result, _ = benchmark.benchmark_function( | |
| row_major_sum, args=(A,), name="row_major", warmup_runs=2, benchmark_runs=5 | |
| ) | |
| results["row_major"] = result | |
| result, _ = benchmark.benchmark_function( | |
| column_major_sum, | |
| args=(A,), | |
| name="column_major", | |
| warmup_runs=2, | |
| benchmark_runs=5, | |
| ) | |
| results["column_major"] = result | |
| result, _ = benchmark.benchmark_function( | |
| random_access_sum, | |
| args=(A,), | |
| name="random_access", | |
| warmup_runs=2, | |
| benchmark_runs=5, | |
| ) | |
| results["random_access"] = result | |
| if all( | |
| key in results for key in ["row_major", "column_major", "random_access"] | |
| ): | |
| row_time = results["row_major"]["mean_time"] | |
| col_time = results["column_major"]["mean_time"] | |
| rand_time = results["random_access"]["mean_time"] | |
| logger.info(f"Spatial Locality Analysis Results:") | |
| logger.info(f"Row-major access: {row_time*1000:.2f} ms") | |
| logger.info(f"Column-major access: {col_time*1000:.2f} ms") | |
| logger.info(f"Random access: {rand_time*1000:.2f} ms") | |
| logger.info(f"Performance Ratios:") | |
| logger.info(f"Column vs Row: {col_time/row_time:.2f}x slower") | |
| logger.info(f"Random vs Row: {rand_time/row_time:.2f}x slower") | |
| return results | |
| def analyze_temporal_locality(data_size: int = 1024 * 1024) -> Dict[str, Any]: | |
| logger.info(f"Analyzing Temporal Locality (Data size: {data_size} elements)") | |
| data = np.random.randn(data_size).astype(np.float32) | |
| def single_pass(arr): | |
| total = 0.0 | |
| for i in range(len(arr)): | |
| total += arr[i] * arr[i] | |
| return total | |
| def multi_pass_sequential(arr, passes=5): | |
| total = 0.0 | |
| for _ in range(passes): | |
| for i in range(len(arr)): | |
| total += arr[i] * arr[i] | |
| return total | |
| def multi_pass_strided(arr, passes=5, stride=1000): | |
| total = 0.0 | |
| for _ in range(passes): | |
| for i in range(0, len(arr), stride): | |
| total += arr[i] * arr[i] | |
| return total / passes | |
| benchmark = PerformanceBenchmark() | |
| results = {} | |
| result, _ = benchmark.benchmark_function( | |
| single_pass, args=(data,), name="single_pass", benchmark_runs=10 | |
| ) | |
| results["single_pass"] = result | |
| result, _ = benchmark.benchmark_function( | |
| multi_pass_sequential, | |
| args=(data, 3), | |
| name="multi_pass_seq", | |
| benchmark_runs=5, | |
| ) | |
| results["multi_pass_seq"] = result | |
| result, _ = benchmark.benchmark_function( | |
| multi_pass_strided, | |
| args=(data, 3), | |
| name="multi_pass_strided", | |
| benchmark_runs=5, | |
| ) | |
| results["multi_pass_strided"] = result | |
| if all( | |
| key in results | |
| for key in ["single_pass", "multi_pass_seq", "multi_pass_strided"] | |
| ): | |
| single_time = results["single_pass"]["mean_time"] | |
| multi_seq_time = results["multi_pass_seq"]["mean_time"] | |
| multi_stride_time = results["multi_pass_strided"]["mean_time"] | |
| logger.info(f"Temporal Locality Analysis Results:") | |
| logger.info(f"Single pass: {single_time*1000:.2f} ms") | |
| logger.info(f"Multi-pass (seq): {multi_seq_time*1000:.2f} ms") | |
| logger.info(f"Multi-pass (stride): {multi_stride_time*1000:.2f} ms") | |
| logger.info(f"Efficiency Analysis:") | |
| logger.info( | |
| f"Sequential reuse efficiency: {(single_time*3)/multi_seq_time:.2f}" | |
| ) | |
| logger.info( | |
| f"Strided access penalty: {multi_stride_time/single_time:.2f}x" | |
| ) | |
| return results | |
| def cache_simulation_analysis() -> Dict[str, float]: | |
| logger.info("Cache Behavior Simulation") | |
| class SimpleCacheSimulator: | |
| def __init__(self, cache_size_kb=32, line_size_bytes=64): | |
| self.cache_size = cache_size_kb * 1024 | |
| self.line_size = line_size_bytes | |
| self.num_lines = self.cache_size // line_size_bytes | |
| self.cache_tags = {} | |
| self.hits = 0 | |
| self.misses = 0 | |
| def access_address(self, address): | |
| line_address = address // self.line_size | |
| cache_line = line_address % self.num_lines | |
| tag = line_address // self.num_lines | |
| if cache_line in self.cache_tags and self.cache_tags[cache_line] == tag: | |
| self.hits += 1 | |
| return True | |
| else: | |
| self.cache_tags[cache_line] = tag | |
| self.misses += 1 | |
| return False | |
| def get_hit_rate(self): | |
| total = self.hits + self.misses | |
| return self.hits / total if total > 0 else 0 | |
| def reset(self): | |
| self.cache_tags = {} | |
| self.hits = 0 | |
| self.misses = 0 | |
| cache = SimpleCacheSimulator(cache_size_kb=32, line_size_bytes=64) | |
| cache.reset() | |
| array_size = 1024 * 1024 | |
| element_size = 4 | |
| for i in range(0, array_size, element_size): | |
| cache.access_address(i) | |
| sequential_hit_rate = cache.get_hit_rate() | |
| cache.reset() | |
| stride = 1024 | |
| for i in range(0, array_size, stride): | |
| cache.access_address(i) | |
| strided_hit_rate = cache.get_hit_rate() | |
| cache.reset() | |
| np.random.seed(42) | |
| addresses = np.random.randint(0, array_size, size=10000) * element_size | |
| for addr in addresses: | |
| cache.access_address(addr) | |
| random_hit_rate = cache.get_hit_rate() | |
| logger.info(f"Cache Simulation Results:") | |
| logger.info( | |
| f"Sequential access hit rate: {sequential_hit_rate:.3f} ({sequential_hit_rate*100:.1f}%)" | |
| ) | |
| logger.info( | |
| f"Strided access hit rate: {strided_hit_rate:.3f} ({strided_hit_rate*100:.1f}%)" | |
| ) | |
| logger.info( | |
| f"Random access hit rate: {random_hit_rate:.3f} ({random_hit_rate*100:.1f}%)" | |
| ) | |
| return { | |
| "sequential": sequential_hit_rate, | |
| "strided": strided_hit_rate, | |
| "random": random_hit_rate, | |
| } | |
| def run_comprehensive_benchmarks() -> Dict[str, Any]: | |
| logger.info("Starting comprehensive benchmark suite") | |
| results = {} | |
| logger.info("Running memory access pattern analysis...") | |
| results["spatial_locality"] = MemoryAccessAnalyzer.analyze_spatial_locality( | |
| matrix_size=1024 | |
| ) | |
| results["temporal_locality"] = MemoryAccessAnalyzer.analyze_temporal_locality( | |
| data_size=512 * 1024 | |
| ) | |
| results["cache_simulation"] = MemoryAccessAnalyzer.cache_simulation_analysis() | |
| logger.info("Comprehensive benchmark suite completed") | |
| return results | |
Xet Storage Details
- Size:
- 12.3 kB
- Xet hash:
- 907076ac013d7a6bffe229f198c0fc3ecd34da428227d44fac4bcb58b356f35c
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.