tahamajs's picture
download
raw
19.1 kB
#!/usr/bin/env python3
import argparse
import sys
import os
import time
import json
from datetime import datetime
from typing import Dict, List, Any, Optional
import logging
from memory_systems import create_memory_system
from benchmark import MemoryBenchmark, BenchmarkResult
from visualization import MemoryVisualizer
from config import ConfigManager
class MemorySystemsPipeline:
def __init__(self, config_file: Optional[str] = None):
self.config_manager = ConfigManager(config_file)
self.benchmark = MemoryBenchmark(self.config_manager.config.results_dir)
self.visualizer = MemoryVisualizer(
self.config_manager.config.visualizations_dir
)
self.logger = logging.getLogger(__name__)
self.results = []
self.scalability_data = {}
self.pattern_data = {}
self.memory_data = {}
def run_quick_test(self) -> Dict[str, Any]:
self.logger.info("🚀 Running quick test of all memory systems")
test_data = [
("user_001", {"name": "Alice", "age": 25, "role": "engineer"}),
("user_002", {"name": "Bob", "age": 30, "role": "designer"}),
("user_003", {"name": "Carol", "age": 28, "role": "manager"}),
]
systems = [
"sequential",
"associative",
"content_addressable",
"adaptive_lru",
"neural_associative",
"compressed",
"hierarchical",
]
results = {}
for system_type in systems:
self.logger.info(f"🔧 Testing {system_type} system")
try:
system = create_memory_system(system_type)
for key, value in test_data:
system.store(key, value)
retrieved_data = {}
for key, _ in test_data:
result = system.retrieve(key)
if result:
retrieved_data[key] = result
results[system_type] = {
"status": "success",
"size": system.size(),
"operations": system.metrics.total_operations,
"retrieved_count": len(retrieved_data),
"avg_access_time": (
sum(system.metrics.access_times)
/ len(system.metrics.access_times)
if system.metrics.access_times
else 0
),
}
self.logger.info(
f" ✓ {system_type}: {results[system_type]['size']} items, "
f"{results[system_type]['operations']} operations"
)
except Exception as e:
self.logger.error(f" ✗ {system_type}: {str(e)}")
results[system_type] = {"status": "error", "error": str(e)}
return results
def run_comprehensive_benchmark(self) -> Dict[str, Any]:
self.logger.info("📊 Running comprehensive benchmark")
benchmark_config = self.config_manager.config.benchmark
all_results = self.benchmark.benchmark_all_systems(
dataset_sizes=benchmark_config.dataset_sizes,
data_types=benchmark_config.data_types,
operations=benchmark_config.operations,
)
self.results = []
for system_results in all_results.values():
self.results.extend(system_results)
results_file = self.benchmark.save_results()
self.logger.info(f"✅ Benchmark completed: {len(self.results)} tests")
return {
"status": "success",
"total_tests": len(self.results),
"results_file": results_file,
"systems_tested": len(all_results),
}
def run_scalability_analysis(self, systems: List[str] = None) -> Dict[str, Any]:
if systems is None:
systems = ["sequential", "associative", "adaptive_lru"]
self.logger.info(f"📈 Running scalability analysis for: {systems}")
benchmark_config = self.config_manager.config.benchmark
for system_type in systems:
self.logger.info(f"🔧 Analyzing scalability of {system_type}")
scalability_results = self.benchmark.scalability_analysis(
system_type=system_type,
max_size=benchmark_config.max_size_scalability,
step_size=benchmark_config.step_size_scalability,
)
serializable_results = []
for result in scalability_results:
serializable_results.append(
{
"dataset_size": result.dataset_size,
"operation": result.operation,
"execution_time": result.execution_time,
"throughput": result.throughput,
"latency": result.latency,
"memory_usage": result.memory_usage,
}
)
self.scalability_data[system_type] = serializable_results
self.logger.info("✅ Scalability analysis completed")
return {
"status": "success",
"systems_analyzed": len(systems),
"scalability_data": self.scalability_data,
}
def run_workload_pattern_analysis(
self, systems: List[str] = None
) -> Dict[str, Any]:
if systems is None:
systems = ["sequential", "associative", "adaptive_lru"]
self.logger.info(f"🎯 Running workload pattern analysis for: {systems}")
benchmark_config = self.config_manager.config.benchmark
for system_type in systems:
self.logger.info(f"🔧 Analyzing workload patterns for {system_type}")
pattern_results = self.benchmark.workload_pattern_analysis(
system_type=system_type, size=benchmark_config.workload_pattern_size
)
serializable_patterns = {}
for pattern_name, results in pattern_results.items():
serializable_patterns[pattern_name] = []
for result in results:
serializable_patterns[pattern_name].append(
{
"execution_time": result.execution_time,
"throughput": result.throughput,
"latency": result.latency,
"memory_usage": result.memory_usage,
}
)
self.pattern_data[system_type] = serializable_patterns
self.logger.info("✅ Workload pattern analysis completed")
return {
"status": "success",
"systems_analyzed": len(systems),
"pattern_data": self.pattern_data,
}
def run_memory_usage_analysis(self) -> Dict[str, Any]:
self.logger.info("💾 Running memory usage analysis")
systems = ["sequential", "associative", "adaptive_lru", "compressed"]
for system_type in systems:
self.logger.info(f"🔧 Analyzing memory usage for {system_type}")
system = create_memory_system(system_type)
test_data = self.benchmark.generate_test_data(1000, "mixed")
start_memory = sys.getsizeof(system)
for key, value in test_data:
system.store(key, value)
end_memory = sys.getsizeof(system)
total_operations = system.metrics.total_operations
memory_used = end_memory - start_memory
efficiency = total_operations / memory_used if memory_used > 0 else 0
self.memory_data[system_type] = {
"total_usage": end_memory,
"usage_increase": memory_used,
"efficiency": efficiency,
"operations": total_operations,
}
self.logger.info("✅ Memory usage analysis completed")
return {
"status": "success",
"systems_analyzed": len(systems),
"memory_data": self.memory_data,
}
def generate_visualizations(self) -> Dict[str, Any]:
self.logger.info("📊 Generating visualizations")
if not self.results:
self.logger.warning("No benchmark results available for visualization")
return {"status": "warning", "message": "No results to visualize"}
viz_data = []
for result in self.results:
viz_data.append(
{
"system_name": result.system_name,
"operation": result.operation,
"dataset_size": result.dataset_size,
"execution_time": result.execution_time,
"throughput": result.throughput,
"latency": result.latency,
"memory_usage": result.memory_usage,
"success_rate": result.success_rate,
}
)
generated_files = self.visualizer.create_comprehensive_report(
results_data=viz_data,
scalability_data=self.scalability_data,
pattern_data=self.pattern_data,
memory_data=self.memory_data,
)
self.logger.info(f"✅ Visualizations generated: {len(generated_files)} files")
return {
"status": "success",
"files_generated": len(generated_files),
"files": generated_files,
}
def generate_report(self) -> Dict[str, Any]:
self.logger.info("📋 Generating comprehensive report")
if self.results:
df_data = []
for result in self.results:
df_data.append(
{
"system": result.system_name,
"operation": result.operation,
"size": result.dataset_size,
"time": result.execution_time,
"throughput": result.throughput,
"latency": result.latency,
"memory": result.memory_usage,
"success_rate": result.success_rate,
}
)
summary_stats = {}
systems = set(r.system_name for r in self.results)
for system in systems:
system_results = [r for r in self.results if r.system_name == system]
summary_stats[system] = {
"total_tests": len(system_results),
"avg_throughput": sum(r.throughput for r in system_results)
/ len(system_results),
"avg_latency": sum(r.latency for r in system_results)
/ len(system_results),
"avg_memory_usage": sum(r.memory_usage for r in system_results)
/ len(system_results),
"avg_success_rate": sum(r.success_rate for r in system_results)
/ len(system_results),
}
else:
summary_stats = {}
report = {
"project_info": {
"name": self.config_manager.config.project_name,
"version": self.config_manager.config.version,
"author": self.config_manager.config.author,
"description": self.config_manager.config.description,
"timestamp": datetime.now().isoformat(),
},
"configuration": {
"benchmark": {
"dataset_sizes": self.config_manager.config.benchmark.dataset_sizes,
"data_types": self.config_manager.config.benchmark.data_types,
"operations": self.config_manager.config.benchmark.operations,
"iterations": self.config_manager.config.benchmark.iterations,
},
"memory_systems": {
"initial_capacity": self.config_manager.config.memory_systems.initial_capacity,
"max_capacity": self.config_manager.config.memory_systems.max_capacity,
"load_factor_threshold": self.config_manager.config.memory_systems.load_factor_threshold,
"similarity_threshold": self.config_manager.config.memory_systems.similarity_threshold,
},
},
"results_summary": {
"total_tests": len(self.results),
"systems_tested": (
len(set(r.system_name for r in self.results)) if self.results else 0
),
"date_range": {
"start": (
min(r.timestamp for r in self.results).isoformat()
if self.results
else None
),
"end": (
max(r.timestamp for r in self.results).isoformat()
if self.results
else None
),
},
},
"performance_summary": summary_stats,
"scalability_analysis": self.scalability_data,
"workload_pattern_analysis": self.pattern_data,
"memory_usage_analysis": self.memory_data,
}
report_file = os.path.join(
self.config_manager.config.results_dir,
f"comprehensive_report_{datetime.now().strftime('%Y%m%d_%H%M%S')}.json",
)
with open(report_file, "w") as f:
json.dump(report, f, indent=2)
self.logger.info(f"✅ Report generated: {report_file}")
return {"status": "success", "report_file": report_file, "summary": report}
def run_full_pipeline(self) -> Dict[str, Any]:
self.logger.info("🚀 Starting full memory systems analysis pipeline")
start_time = time.time()
pipeline_results = {}
try:
self.logger.info("Step 1/7: Quick system test")
pipeline_results["quick_test"] = self.run_quick_test()
self.logger.info("Step 2/7: Comprehensive benchmark")
pipeline_results["benchmark"] = self.run_comprehensive_benchmark()
self.logger.info("Step 3/7: Scalability analysis")
pipeline_results["scalability"] = self.run_scalability_analysis()
self.logger.info("Step 4/7: Workload pattern analysis")
pipeline_results["workload_patterns"] = self.run_workload_pattern_analysis()
self.logger.info("Step 5/7: Memory usage analysis")
pipeline_results["memory_usage"] = self.run_memory_usage_analysis()
self.logger.info("Step 6/7: Generate visualizations")
pipeline_results["visualizations"] = self.generate_visualizations()
self.logger.info("Step 7/7: Generate comprehensive report")
pipeline_results["report"] = self.generate_report()
total_time = time.time() - start_time
pipeline_results["pipeline_summary"] = {
"status": "success",
"total_time": total_time,
"steps_completed": 7,
}
self.logger.info(f"✅ Full pipeline completed in {total_time:.2f} seconds")
except Exception as e:
self.logger.error(f"❌ Pipeline failed: {str(e)}")
pipeline_results["pipeline_summary"] = {
"status": "error",
"error": str(e),
"total_time": time.time() - start_time,
}
return pipeline_results
def main():
parser = argparse.ArgumentParser(
description="Memory Systems Analysis Pipeline",
formatter_class=argparse.RawDescriptionHelpFormatter,
epilog=
)
parser.add_argument("--config", "-c", type=str, help="Configuration file path")
parser.add_argument("--quick-test", action="store_true", help="Run quick test only")
parser.add_argument(
"--benchmark", action="store_true", help="Run comprehensive benchmark"
)
parser.add_argument(
"--scalability", action="store_true", help="Run scalability analysis"
)
parser.add_argument(
"--workload-patterns", action="store_true", help="Run workload pattern analysis"
)
parser.add_argument(
"--memory-usage", action="store_true", help="Run memory usage analysis"
)
parser.add_argument(
"--visualizations", action="store_true", help="Generate visualizations"
)
parser.add_argument(
"--report", action="store_true", help="Generate comprehensive report"
)
parser.add_argument(
"--full-pipeline", action="store_true", help="Run complete analysis pipeline"
)
parser.add_argument("--systems", nargs="+", help="Specific systems to test")
parser.add_argument(
"--verbose", "-v", action="store_true", help="Enable verbose logging"
)
args = parser.parse_args()
if args.verbose:
logging.getLogger().setLevel(logging.DEBUG)
pipeline = MemorySystemsPipeline(args.config)
print("🧠 Memory Systems Analysis Pipeline")
print("=" * 50)
pipeline.config_manager.print_config_summary()
issues = pipeline.config_manager.validate_config()
if issues:
print("\n⚠️ Configuration Issues:")
for issue in issues:
print(f" - {issue}")
print("Continuing with default values...")
print()
if args.full_pipeline:
results = pipeline.run_full_pipeline()
elif args.quick_test:
results = pipeline.run_quick_test()
elif args.benchmark:
results = pipeline.run_comprehensive_benchmark()
elif args.scalability:
systems = args.systems or ["sequential", "associative", "adaptive_lru"]
results = pipeline.run_scalability_analysis(systems)
elif args.workload_patterns:
systems = args.systems or ["sequential", "associative", "adaptive_lru"]
results = pipeline.run_workload_pattern_analysis(systems)
elif args.memory_usage:
results = pipeline.run_memory_usage_analysis()
elif args.visualizations:
results = pipeline.generate_visualizations()
elif args.report:
results = pipeline.generate_report()
else:
print("No specific operation requested, running quick test...")
results = pipeline.run_quick_test()
print("\n📊 Results Summary:")
print("-" * 30)
if isinstance(results, dict):
if "status" in results:
print(f"Status: {results['status']}")
if "total_tests" in results:
print(f"Total tests: {results['total_tests']}")
if "files_generated" in results:
print(f"Files generated: {results['files_generated']}")
if "total_time" in results:
print(f"Total time: {results['total_time']:.2f} seconds")
print("\n✅ Analysis completed!")
return 0
if __name__ == "__main__":
sys.exit(main())

Xet Storage Details

Size:
19.1 kB
·
Xet hash:
2ece501c68d4fb4b20c12f1d53076a7296f6dc6675e3aa499d7769513a6c9fc8

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.