tahamajs's picture
download
raw
12.3 kB
import time
import gc
import torch
import numpy as np
import pandas as pd
from typing import Dict, List, Any, Callable, Optional, Tuple
from collections import defaultdict
import logging
import psutil
import os
logger = logging.getLogger(__name__)
class PerformanceBenchmark:
def __init__(self):
self.results = defaultdict(list)
self.memory_snapshots = []
def benchmark_function(
self,
func: Callable,
args: tuple = (),
kwargs: dict = {},
name: str = "unnamed",
warmup_runs: int = 3,
benchmark_runs: int = 10,
) -> Tuple[Dict[str, Any], Any]:
logger.info(f"Benchmarking {name}")
for _ in range(warmup_runs):
try:
_ = func(*args, **kwargs)
except Exception as e:
logger.error(f"Warmup failed for {name}: {e}")
return {}, None
gc.collect()
if torch.cuda.is_available():
torch.cuda.empty_cache()
torch.cuda.synchronize()
initial_memory = self._get_memory_usage()
times = []
for run in range(benchmark_runs):
start_time = time.perf_counter()
try:
result = func(*args, **kwargs)
if torch.cuda.is_available():
torch.cuda.synchronize()
end_time = time.perf_counter()
times.append(end_time - start_time)
except Exception as e:
logger.error(f"Benchmark run {run} failed for {name}: {e}")
continue
final_memory = self._get_memory_usage()
if not times:
logger.error(f"No successful runs for {name}")
return {}, None
peak_memory = self._get_peak_memory()
benchmark_result = {
"name": name,
"mean_time": np.mean(times),
"std_time": np.std(times),
"min_time": np.min(times),
"max_time": np.max(times),
"peak_cpu_memory": peak_memory["cpu"],
"peak_gpu_memory": peak_memory["gpu"],
"peak_process_memory": peak_memory["process"],
"memory_delta": final_memory["process"] - initial_memory["process"],
}
self.results[name].append(benchmark_result)
return benchmark_result, result
def _get_memory_usage(self) -> Dict[str, float]:
process = psutil.Process()
memory_info = {
"cpu": psutil.virtual_memory().used / (1024**2),
"process": process.memory_info().rss / (1024**2),
"gpu": 0.0,
}
if torch.cuda.is_available():
memory_info["gpu"] = torch.cuda.memory_allocated() / (1024**2)
return memory_info
def _get_peak_memory(self) -> Dict[str, float]:
return self._get_memory_usage()
def compare_results(self) -> Optional[pd.DataFrame]:
if not self.results:
logger.warning("No benchmark results available")
return None
df_data = []
for name, results in self.results.items():
for result in results:
df_data.append(result)
if not df_data:
logger.warning("No benchmark data to compare")
return None
df = pd.DataFrame(df_data)
logger.info("Benchmark Comparison Results:")
logger.info("=" * 80)
logger.info(
f"{'Function':<20} {'Time (ms)':<12} {'Std (ms)':<10} {'Peak GPU (MB)':<15} {'Peak CPU (MB)':<15}"
)
logger.info("-" * 80)
for _, row in df.iterrows():
logger.info(
f"{row['name']:<20} {row['mean_time']*1000:<12.3f} {row['std_time']*1000:<10.3f} "
f"{row['peak_gpu_memory']:<15.1f} {row['peak_cpu_memory']:<15.1f}"
)
return df
def clear_results(self):
self.results.clear()
self.memory_snapshots.clear()
class MemoryAccessAnalyzer:
@staticmethod
def analyze_spatial_locality(matrix_size: int = 2048) -> Dict[str, Any]:
logger.info(
f"Analyzing Spatial Locality (Matrix size: {matrix_size}x{matrix_size})"
)
A = np.random.randn(matrix_size, matrix_size).astype(np.float32)
def row_major_sum(matrix):
total = 0.0
for i in range(matrix.shape[0]):
for j in range(matrix.shape[1]):
total += matrix[i, j]
return total
def column_major_sum(matrix):
total = 0.0
for j in range(matrix.shape[1]):
for i in range(matrix.shape[0]):
total += matrix[i, j]
return total
def random_access_sum(matrix):
indices = np.random.permutation(matrix_size * matrix_size)
total = 0.0
for idx in indices:
i, j = divmod(idx, matrix_size)
total += matrix[i, j]
return total
benchmark = PerformanceBenchmark()
results = {}
result, _ = benchmark.benchmark_function(
row_major_sum, args=(A,), name="row_major", warmup_runs=2, benchmark_runs=5
)
results["row_major"] = result
result, _ = benchmark.benchmark_function(
column_major_sum,
args=(A,),
name="column_major",
warmup_runs=2,
benchmark_runs=5,
)
results["column_major"] = result
result, _ = benchmark.benchmark_function(
random_access_sum,
args=(A,),
name="random_access",
warmup_runs=2,
benchmark_runs=5,
)
results["random_access"] = result
if all(
key in results for key in ["row_major", "column_major", "random_access"]
):
row_time = results["row_major"]["mean_time"]
col_time = results["column_major"]["mean_time"]
rand_time = results["random_access"]["mean_time"]
logger.info(f"Spatial Locality Analysis Results:")
logger.info(f"Row-major access: {row_time*1000:.2f} ms")
logger.info(f"Column-major access: {col_time*1000:.2f} ms")
logger.info(f"Random access: {rand_time*1000:.2f} ms")
logger.info(f"Performance Ratios:")
logger.info(f"Column vs Row: {col_time/row_time:.2f}x slower")
logger.info(f"Random vs Row: {rand_time/row_time:.2f}x slower")
return results
@staticmethod
def analyze_temporal_locality(data_size: int = 1024 * 1024) -> Dict[str, Any]:
logger.info(f"Analyzing Temporal Locality (Data size: {data_size} elements)")
data = np.random.randn(data_size).astype(np.float32)
def single_pass(arr):
total = 0.0
for i in range(len(arr)):
total += arr[i] * arr[i]
return total
def multi_pass_sequential(arr, passes=5):
total = 0.0
for _ in range(passes):
for i in range(len(arr)):
total += arr[i] * arr[i]
return total
def multi_pass_strided(arr, passes=5, stride=1000):
total = 0.0
for _ in range(passes):
for i in range(0, len(arr), stride):
total += arr[i] * arr[i]
return total / passes
benchmark = PerformanceBenchmark()
results = {}
result, _ = benchmark.benchmark_function(
single_pass, args=(data,), name="single_pass", benchmark_runs=10
)
results["single_pass"] = result
result, _ = benchmark.benchmark_function(
multi_pass_sequential,
args=(data, 3),
name="multi_pass_seq",
benchmark_runs=5,
)
results["multi_pass_seq"] = result
result, _ = benchmark.benchmark_function(
multi_pass_strided,
args=(data, 3),
name="multi_pass_strided",
benchmark_runs=5,
)
results["multi_pass_strided"] = result
if all(
key in results
for key in ["single_pass", "multi_pass_seq", "multi_pass_strided"]
):
single_time = results["single_pass"]["mean_time"]
multi_seq_time = results["multi_pass_seq"]["mean_time"]
multi_stride_time = results["multi_pass_strided"]["mean_time"]
logger.info(f"Temporal Locality Analysis Results:")
logger.info(f"Single pass: {single_time*1000:.2f} ms")
logger.info(f"Multi-pass (seq): {multi_seq_time*1000:.2f} ms")
logger.info(f"Multi-pass (stride): {multi_stride_time*1000:.2f} ms")
logger.info(f"Efficiency Analysis:")
logger.info(
f"Sequential reuse efficiency: {(single_time*3)/multi_seq_time:.2f}"
)
logger.info(
f"Strided access penalty: {multi_stride_time/single_time:.2f}x"
)
return results
@staticmethod
def cache_simulation_analysis() -> Dict[str, float]:
logger.info("Cache Behavior Simulation")
class SimpleCacheSimulator:
def __init__(self, cache_size_kb=32, line_size_bytes=64):
self.cache_size = cache_size_kb * 1024
self.line_size = line_size_bytes
self.num_lines = self.cache_size // line_size_bytes
self.cache_tags = {}
self.hits = 0
self.misses = 0
def access_address(self, address):
line_address = address // self.line_size
cache_line = line_address % self.num_lines
tag = line_address // self.num_lines
if cache_line in self.cache_tags and self.cache_tags[cache_line] == tag:
self.hits += 1
return True
else:
self.cache_tags[cache_line] = tag
self.misses += 1
return False
def get_hit_rate(self):
total = self.hits + self.misses
return self.hits / total if total > 0 else 0
def reset(self):
self.cache_tags = {}
self.hits = 0
self.misses = 0
cache = SimpleCacheSimulator(cache_size_kb=32, line_size_bytes=64)
cache.reset()
array_size = 1024 * 1024
element_size = 4
for i in range(0, array_size, element_size):
cache.access_address(i)
sequential_hit_rate = cache.get_hit_rate()
cache.reset()
stride = 1024
for i in range(0, array_size, stride):
cache.access_address(i)
strided_hit_rate = cache.get_hit_rate()
cache.reset()
np.random.seed(42)
addresses = np.random.randint(0, array_size, size=10000) * element_size
for addr in addresses:
cache.access_address(addr)
random_hit_rate = cache.get_hit_rate()
logger.info(f"Cache Simulation Results:")
logger.info(
f"Sequential access hit rate: {sequential_hit_rate:.3f} ({sequential_hit_rate*100:.1f}%)"
)
logger.info(
f"Strided access hit rate: {strided_hit_rate:.3f} ({strided_hit_rate*100:.1f}%)"
)
logger.info(
f"Random access hit rate: {random_hit_rate:.3f} ({random_hit_rate*100:.1f}%)"
)
return {
"sequential": sequential_hit_rate,
"strided": strided_hit_rate,
"random": random_hit_rate,
}
def run_comprehensive_benchmarks() -> Dict[str, Any]:
logger.info("Starting comprehensive benchmark suite")
results = {}
logger.info("Running memory access pattern analysis...")
results["spatial_locality"] = MemoryAccessAnalyzer.analyze_spatial_locality(
matrix_size=1024
)
results["temporal_locality"] = MemoryAccessAnalyzer.analyze_temporal_locality(
data_size=512 * 1024
)
results["cache_simulation"] = MemoryAccessAnalyzer.cache_simulation_analysis()
logger.info("Comprehensive benchmark suite completed")
return results

Xet Storage Details

Size:
12.3 kB
·
Xet hash:
907076ac013d7a6bffe229f198c0fc3ecd34da428227d44fac4bcb58b356f35c

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.