Buckets:
| #!/usr/bin/env python3 | |
| import time | |
| import random | |
| import numpy as np | |
| import pandas as pd | |
| import matplotlib.pyplot as plt | |
| import seaborn as sns | |
| from typing import Dict, List, Tuple, Any, Optional | |
| from dataclasses import dataclass | |
| import json | |
| import os | |
| from datetime import datetime | |
| import psutil | |
| import sys | |
| from memory_systems import ( | |
| SequentialMemory, | |
| AssociativeMemory, | |
| ContentAddressableMemory, | |
| AdaptiveLRUCache, | |
| NeuralAssociativeMemory, | |
| CompressedMemorySystem, | |
| HierarchicalMemorySystem, | |
| create_memory_system, | |
| ) | |
| class BenchmarkResult: | |
| system_name: str | |
| operation: str | |
| dataset_size: int | |
| execution_time: float | |
| memory_usage: int | |
| success_rate: float | |
| throughput: float | |
| latency: float | |
| timestamp: datetime | |
| class MemoryBenchmark: | |
| def __init__(self, results_dir: str = "benchmark_results"): | |
| self.results_dir = results_dir | |
| self.results = [] | |
| self._ensure_results_dir() | |
| def _ensure_results_dir(self): | |
| if not os.path.exists(self.results_dir): | |
| os.makedirs(self.results_dir) | |
| def generate_test_data( | |
| self, size: int, data_type: str = "mixed" | |
| ) -> List[Tuple[Any, Any]]: | |
| data = [] | |
| if data_type == "mixed": | |
| for i in range(size): | |
| key = f"key_{i:06d}" | |
| value = { | |
| "id": i, | |
| "name": f"Item_{i}", | |
| "value": random.uniform(0, 1000), | |
| "category": random.choice(["A", "B", "C", "D"]), | |
| "timestamp": time.time(), | |
| "metadata": { | |
| "source": random.choice(["user", "system", "api"]), | |
| "priority": random.randint(1, 10), | |
| "tags": [f"tag_{j}" for j in range(random.randint(1, 5))], | |
| }, | |
| } | |
| data.append((key, value)) | |
| elif data_type == "simple": | |
| for i in range(size): | |
| key = f"key_{i:06d}" | |
| value = f"value_{i}" | |
| data.append((key, value)) | |
| elif data_type == "numeric": | |
| for i in range(size): | |
| key = i | |
| value = [random.uniform(0, 1) for _ in range(10)] | |
| data.append((key, value)) | |
| elif data_type == "text": | |
| words = ["apple", "banana", "cherry", "date", "elderberry", "fig", "grape"] | |
| for i in range(size): | |
| key = f"doc_{i:06d}" | |
| value = " ".join(random.choices(words, k=random.randint(5, 20))) | |
| data.append((key, value)) | |
| return data | |
| def benchmark_operation( | |
| self, | |
| system, | |
| operation: str, | |
| test_data: List[Tuple[Any, Any]], | |
| iterations: int = 1, | |
| ) -> BenchmarkResult: | |
| start_time = time.time() | |
| start_memory = psutil.Process().memory_info().rss | |
| success_count = 0 | |
| total_operations = 0 | |
| for iteration in range(iterations): | |
| if operation == "store": | |
| for key, value in test_data: | |
| if system.store(key, value): | |
| success_count += 1 | |
| total_operations += 1 | |
| elif operation == "retrieve": | |
| for key, _ in test_data: | |
| result = system.retrieve(key) | |
| if result is not None: | |
| success_count += 1 | |
| total_operations += 1 | |
| elif operation == "mixed": | |
| for i, (key, value) in enumerate(test_data): | |
| if i % 10 < 7: | |
| result = system.retrieve(key) | |
| if result is not None: | |
| success_count += 1 | |
| elif i % 10 < 9: | |
| if system.store(f"new_{key}", value): | |
| success_count += 1 | |
| else: | |
| if system.delete(key): | |
| success_count += 1 | |
| total_operations += 1 | |
| end_time = time.time() | |
| end_memory = psutil.Process().memory_info().rss | |
| execution_time = end_time - start_time | |
| memory_usage = end_memory - start_memory | |
| success_rate = success_count / total_operations if total_operations > 0 else 0 | |
| throughput = total_operations / execution_time if execution_time > 0 else 0 | |
| latency = execution_time / total_operations if total_operations > 0 else 0 | |
| return BenchmarkResult( | |
| system_name=system.name, | |
| operation=operation, | |
| dataset_size=len(test_data), | |
| execution_time=execution_time, | |
| memory_usage=memory_usage, | |
| success_rate=success_rate, | |
| throughput=throughput, | |
| latency=latency, | |
| timestamp=datetime.now(), | |
| ) | |
| def benchmark_system( | |
| self, | |
| system_type: str, | |
| dataset_sizes: List[int], | |
| data_types: List[str] = ["mixed"], | |
| operations: List[str] = ["store", "retrieve", "mixed"], | |
| ) -> List[BenchmarkResult]: | |
| results = [] | |
| for data_type in data_types: | |
| for size in dataset_sizes: | |
| print( | |
| f"๐ง Benchmarking {system_type} with {data_type} data (size: {size})" | |
| ) | |
| system = create_memory_system(system_type) | |
| test_data = self.generate_test_data(size, data_type) | |
| for operation in operations: | |
| if operation == "store": | |
| for key, value in test_data[: size // 2]: | |
| system.store(key, value) | |
| result = self.benchmark_operation(system, operation, test_data) | |
| results.append(result) | |
| self.results.append(result) | |
| print( | |
| f" โ {operation}: {result.execution_time:.4f}s, " | |
| f"throughput: {result.throughput:.2f} ops/s" | |
| ) | |
| return results | |
| def benchmark_all_systems( | |
| self, | |
| dataset_sizes: List[int] = [100, 500, 1000, 2000], | |
| data_types: List[str] = ["mixed", "simple", "numeric", "text"], | |
| operations: List[str] = ["store", "retrieve", "mixed"], | |
| ) -> Dict[str, List[BenchmarkResult]]: | |
| system_types = [ | |
| "sequential", | |
| "associative", | |
| "content_addressable", | |
| "adaptive_lru", | |
| "neural_associative", | |
| "compressed", | |
| "hierarchical", | |
| ] | |
| all_results = {} | |
| print("๐ Starting comprehensive memory systems benchmark") | |
| print("=" * 60) | |
| for system_type in system_types: | |
| print(f"\n๐ Benchmarking {system_type} system...") | |
| results = self.benchmark_system( | |
| system_type, dataset_sizes, data_types, operations | |
| ) | |
| all_results[system_type] = results | |
| print("\nโ Benchmark completed!") | |
| return all_results | |
| def scalability_analysis( | |
| self, system_type: str, max_size: int = 5000, step_size: int = 500 | |
| ) -> List[BenchmarkResult]: | |
| sizes = list(range(step_size, max_size + 1, step_size)) | |
| results = [] | |
| print(f"๐ Scalability analysis for {system_type}") | |
| print("-" * 40) | |
| for size in sizes: | |
| print(f" Testing size: {size}") | |
| system = create_memory_system(system_type) | |
| test_data = self.generate_test_data(size, "mixed") | |
| store_result = self.benchmark_operation(system, "store", test_data) | |
| results.append(store_result) | |
| retrieve_result = self.benchmark_operation(system, "retrieve", test_data) | |
| results.append(retrieve_result) | |
| print( | |
| f" Store: {store_result.execution_time:.4f}s, " | |
| f"Retrieve: {retrieve_result.execution_time:.4f}s" | |
| ) | |
| return results | |
| def workload_pattern_analysis( | |
| self, system_type: str, size: int = 1000 | |
| ) -> Dict[str, List[BenchmarkResult]]: | |
| patterns = { | |
| "uniform": "Uniform random access", | |
| "sequential": "Sequential access", | |
| "skewed": "80-20 skewed access", | |
| "burst": "Burst access patterns", | |
| } | |
| results = {} | |
| print(f"๐ฏ Workload pattern analysis for {system_type}") | |
| print("-" * 50) | |
| for pattern_name, description in patterns.items(): | |
| print(f" Testing {description}...") | |
| system = create_memory_system(system_type) | |
| test_data = self.generate_test_data(size, "mixed") | |
| for key, value in test_data: | |
| system.store(key, value) | |
| pattern_results = [] | |
| if pattern_name == "uniform": | |
| random.shuffle(test_data) | |
| result = self.benchmark_operation(system, "retrieve", test_data) | |
| pattern_results.append(result) | |
| elif pattern_name == "sequential": | |
| result = self.benchmark_operation(system, "retrieve", test_data) | |
| pattern_results.append(result) | |
| elif pattern_name == "skewed": | |
| hot_data = test_data[: size // 5] | |
| cold_data = test_data[size // 5 :] | |
| skewed_data = [] | |
| for _ in range(size): | |
| if random.random() < 0.8: | |
| skewed_data.append(random.choice(hot_data)) | |
| else: | |
| skewed_data.append(random.choice(cold_data)) | |
| result = self.benchmark_operation(system, "retrieve", skewed_data) | |
| pattern_results.append(result) | |
| elif pattern_name == "burst": | |
| burst_data = [] | |
| for _ in range(10): | |
| for _ in range(size // 10): | |
| burst_data.append(random.choice(test_data)) | |
| for _ in range(size // 20): | |
| burst_data.append(random.choice(test_data)) | |
| result = self.benchmark_operation(system, "retrieve", burst_data) | |
| pattern_results.append(result) | |
| results[pattern_name] = pattern_results | |
| print(f" {description}: {pattern_results[0].execution_time:.4f}s") | |
| return results | |
| def save_results(self, filename: str = None): | |
| if filename is None: | |
| timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") | |
| filename = f"benchmark_results_{timestamp}.json" | |
| filepath = os.path.join(self.results_dir, filename) | |
| serializable_results = [] | |
| for result in self.results: | |
| serializable_results.append( | |
| { | |
| "system_name": result.system_name, | |
| "operation": result.operation, | |
| "dataset_size": result.dataset_size, | |
| "execution_time": result.execution_time, | |
| "memory_usage": result.memory_usage, | |
| "success_rate": result.success_rate, | |
| "throughput": result.throughput, | |
| "latency": result.latency, | |
| "timestamp": result.timestamp.isoformat(), | |
| } | |
| ) | |
| with open(filepath, "w") as f: | |
| json.dump(serializable_results, f, indent=2) | |
| print(f"๐พ Results saved to {filepath}") | |
| return filepath | |
| def load_results(self, filename: str) -> List[BenchmarkResult]: | |
| filepath = os.path.join(self.results_dir, filename) | |
| with open(filepath, "r") as f: | |
| data = json.load(f) | |
| results = [] | |
| for item in data: | |
| result = BenchmarkResult( | |
| system_name=item["system_name"], | |
| operation=item["operation"], | |
| dataset_size=item["dataset_size"], | |
| execution_time=item["execution_time"], | |
| memory_usage=item["memory_usage"], | |
| success_rate=item["success_rate"], | |
| throughput=item["throughput"], | |
| latency=item["latency"], | |
| timestamp=datetime.fromisoformat(item["timestamp"]), | |
| ) | |
| results.append(result) | |
| return results | |
| def generate_summary_report(self) -> Dict[str, Any]: | |
| if not self.results: | |
| return {"error": "No benchmark results available"} | |
| df = pd.DataFrame( | |
| [ | |
| { | |
| "system": r.system_name, | |
| "operation": r.operation, | |
| "size": r.dataset_size, | |
| "time": r.execution_time, | |
| "throughput": r.throughput, | |
| "latency": r.latency, | |
| "memory": r.memory_usage, | |
| "success_rate": r.success_rate, | |
| } | |
| for r in self.results | |
| ] | |
| ) | |
| summary = { | |
| "total_tests": len(self.results), | |
| "systems_tested": df["system"].nunique(), | |
| "operations_tested": df["operation"].nunique(), | |
| "date_range": { | |
| "start": min(r.timestamp for r in self.results).isoformat(), | |
| "end": max(r.timestamp for r in self.results).isoformat(), | |
| }, | |
| "performance_summary": {}, | |
| } | |
| for system in df["system"].unique(): | |
| system_data = df[df["system"] == system] | |
| summary["performance_summary"][system] = { | |
| "avg_throughput": system_data["throughput"].mean(), | |
| "avg_latency": system_data["latency"].mean(), | |
| "avg_memory_usage": system_data["memory"].mean(), | |
| "avg_success_rate": system_data["success_rate"].mean(), | |
| "total_tests": len(system_data), | |
| } | |
| return summary | |
| def run_comprehensive_benchmark(): | |
| benchmark = MemoryBenchmark() | |
| print("๐ง Memory Systems Comprehensive Benchmark") | |
| print("=" * 60) | |
| print("This benchmark will test all memory systems across different scenarios") | |
| print("Estimated time: 10-15 minutes") | |
| print() | |
| dataset_sizes = [100, 500, 1000] | |
| data_types = ["mixed", "simple"] | |
| operations = ["store", "retrieve", "mixed"] | |
| all_results = benchmark.benchmark_all_systems(dataset_sizes, data_types, operations) | |
| print("\n๐ Running scalability analysis...") | |
| key_systems = ["sequential", "associative", "adaptive_lru"] | |
| for system_type in key_systems: | |
| scalability_results = benchmark.scalability_analysis( | |
| system_type, max_size=2000, step_size=500 | |
| ) | |
| benchmark.results.extend(scalability_results) | |
| print("\n๐ฏ Running workload pattern analysis...") | |
| for system_type in key_systems: | |
| pattern_results = benchmark.workload_pattern_analysis(system_type, size=1000) | |
| for pattern_name, results in pattern_results.items(): | |
| benchmark.results.extend(results) | |
| results_file = benchmark.save_results() | |
| summary = benchmark.generate_summary_report() | |
| summary_file = os.path.join(benchmark.results_dir, "summary_report.json") | |
| with open(summary_file, "w") as f: | |
| json.dump(summary, f, indent=2) | |
| print(f"\n๐ Summary report saved to {summary_file}") | |
| print(f"๐ Total tests completed: {summary['total_tests']}") | |
| print(f"๐ง Systems tested: {summary['systems_tested']}") | |
| return benchmark, results_file, summary_file | |
| if __name__ == "__main__": | |
| benchmark, results_file, summary_file = run_comprehensive_benchmark() | |
| print("\nโ Benchmark completed successfully!") | |
| print(f"๐ Results directory: {benchmark.results_dir}") | |
| print(f"๐ Detailed results: {results_file}") | |
| print(f"๐ Summary report: {summary_file}") | |
Xet Storage Details
- Size:
- 15.9 kB
- Xet hash:
- 08ed6529e8ad7a837d29114d4cd89d8136c3a744fd92e889aff767ddf937ecb2
ยท
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.