Buckets:
| #!/usr/bin/env python3 | |
| import numpy as np | |
| import pandas as pd | |
| import matplotlib.pyplot as plt | |
| import seaborn as sns | |
| from scipy import stats | |
| from scipy.cluster.hierarchy import dendrogram, linkage | |
| from sklearn.cluster import KMeans, DBSCAN, AgglomerativeClustering | |
| from sklearn.decomposition import PCA, t-SNE, UMAP | |
| from sklearn.manifold import TSNE | |
| from sklearn.preprocessing import StandardScaler | |
| from sklearn.model_selection import train_test_split | |
| from sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor | |
| from sklearn.linear_model import LinearRegression, Ridge, Lasso | |
| from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error | |
| from sklearn.neighbors import LocalOutlierFactor | |
| from sklearn.svm import OneClassSVM | |
| import warnings | |
| warnings.filterwarnings('ignore') | |
| from typing import Dict, List, Tuple, Any, Optional, Union | |
| import json | |
| import pickle | |
| from dataclasses import dataclass | |
| from datetime import datetime, timedelta | |
| import plotly.graph_objects as go | |
| import plotly.express as px | |
| from plotly.subplots import make_subplots | |
| class StatisticalResults: | |
| mean: float = 0.0 | |
| median: float = 0.0 | |
| std: float = 0.0 | |
| skewness: float = 0.0 | |
| kurtosis: float = 0.0 | |
| correlation_matrix: np.ndarray = None | |
| significant_correlations: List[Tuple[str, str, float]] = None | |
| cluster_labels: np.ndarray = None | |
| silhouette_score: float = 0.0 | |
| inertia: float = 0.0 | |
| explained_variance_ratio: np.ndarray = None | |
| cumulative_variance: np.ndarray = None | |
| trend_coefficient: float = 0.0 | |
| seasonality_strength: float = 0.0 | |
| autocorrelation_lags: Dict[int, float] = None | |
| model_performance: Dict[str, float] = None | |
| feature_importance: Dict[str, float] = None | |
| anomaly_scores: np.ndarray = None | |
| anomaly_indices: List[int] = None | |
| def __post_init__(self): | |
| if self.correlation_matrix is None: | |
| self.correlation_matrix = np.array([]) | |
| if self.significant_correlations is None: | |
| self.significant_correlations = [] | |
| if self.cluster_labels is None: | |
| self.cluster_labels = np.array([]) | |
| if self.explained_variance_ratio is None: | |
| self.explained_variance_ratio = np.array([]) | |
| if self.cumulative_variance is None: | |
| self.cumulative_variance = np.array([]) | |
| if self.autocorrelation_lags is None: | |
| self.autocorrelation_lags = {} | |
| if self.model_performance is None: | |
| self.model_performance = {} | |
| if self.feature_importance is None: | |
| self.feature_importance = {} | |
| if self.anomaly_scores is None: | |
| self.anomaly_scores = np.array([]) | |
| if self.anomaly_indices is None: | |
| self.anomaly_indices = [] | |
| class BayesianAnalyzer: | |
| def __init__(self): | |
| self.prior_distributions = {} | |
| self.posterior_samples = {} | |
| def set_prior(self, parameter: str, distribution_type: str, **params): | |
| if distribution_type == "normal": | |
| self.prior_distributions[parameter] = { | |
| 'type': 'normal', | |
| 'mean': params.get('mean', 0), | |
| 'std': params.get('std', 1) | |
| } | |
| elif distribution_type == "beta": | |
| self.prior_distributions[parameter] = { | |
| 'type': 'beta', | |
| 'alpha': params.get('alpha', 1), | |
| 'beta': params.get('beta', 1) | |
| } | |
| elif distribution_type == "gamma": | |
| self.prior_distributions[parameter] = { | |
| 'type': 'gamma', | |
| 'shape': params.get('shape', 1), | |
| 'scale': params.get('scale', 1) | |
| } | |
| def bayesian_inference(self, data: np.ndarray, parameter: str, | |
| n_samples: int = 10000) -> np.ndarray: | |
| if parameter not in self.prior_distributions: | |
| raise ValueError(f"No prior set for parameter: {parameter}") | |
| prior = self.prior_distributions[parameter] | |
| samples = [] | |
| current_value = np.mean(data) | |
| for _ in range(n_samples): | |
| if prior['type'] == 'normal': | |
| proposal = current_value + np.random.normal(0, 0.1) | |
| prior_prob_current = stats.norm.pdf(current_value, | |
| prior['mean'], prior['std']) | |
| prior_prob_proposal = stats.norm.pdf(proposal, | |
| prior['mean'], prior['std']) | |
| else: | |
| proposal = current_value + np.random.normal(0, 0.1) | |
| prior_prob_current = 1.0 | |
| prior_prob_proposal = 1.0 | |
| likelihood_current = np.prod(stats.norm.pdf(data, current_value, np.std(data))) | |
| likelihood_proposal = np.prod(stats.norm.pdf(data, proposal, np.std(data))) | |
| ratio = (likelihood_proposal * prior_prob_proposal) / \ | |
| (likelihood_current * prior_prob_current) | |
| if np.random.random() < min(1, ratio): | |
| current_value = proposal | |
| samples.append(current_value) | |
| self.posterior_samples[parameter] = np.array(samples) | |
| return np.array(samples) | |
| def credible_interval(self, parameter: str, alpha: float = 0.05) -> Tuple[float, float]: | |
| if parameter not in self.posterior_samples: | |
| raise ValueError(f"No posterior samples for parameter: {parameter}") | |
| samples = self.posterior_samples[parameter] | |
| lower = np.percentile(samples, (alpha/2) * 100) | |
| upper = np.percentile(samples, (1 - alpha/2) * 100) | |
| return lower, upper | |
| class TimeSeriesAnalyzer: | |
| def __init__(self): | |
| self.trend_model = None | |
| self.seasonal_model = None | |
| def decompose_timeseries(self, data: np.ndarray, period: int = None) -> Dict[str, np.ndarray]: | |
| n = len(data) | |
| if period is None: | |
| period = max(2, n // 10) | |
| trend = np.convolve(data, np.ones(period)/period, mode='same') | |
| detrended = data - trend | |
| seasonal = np.zeros_like(data) | |
| if period > 1: | |
| for i in range(period): | |
| seasonal[i::period] = np.mean(detrended[i::period]) | |
| residual = data - trend - seasonal | |
| return { | |
| 'trend': trend, | |
| 'seasonal': seasonal, | |
| 'residual': residual, | |
| 'original': data | |
| } | |
| def calculate_autocorrelation(self, data: np.ndarray, max_lags: int = 20) -> Dict[int, float]: | |
| autocorr = {} | |
| n = len(data) | |
| for lag in range(max_lags + 1): | |
| if lag == 0: | |
| autocorr[lag] = 1.0 | |
| else: | |
| numerator = np.sum((data[lag:] - np.mean(data)) * | |
| (data[:-lag] - np.mean(data))) | |
| denominator = np.sum((data - np.mean(data)) ** 2) | |
| if denominator != 0: | |
| autocorr[lag] = numerator / denominator | |
| else: | |
| autocorr[lag] = 0.0 | |
| return autocorr | |
| def detect_trend(self, data: np.ndarray) -> Dict[str, float]: | |
| n = len(data) | |
| x = np.arange(n) | |
| slope, intercept, r_value, p_value, std_err = stats.linregress(x, data) | |
| def mann_kendall_test(y): | |
| n = len(y) | |
| s = 0 | |
| for i in range(n-1): | |
| for j in range(i+1, n): | |
| s += np.sign(y[j] - y[i]) | |
| var_s = n * (n - 1) * (2 * n + 5) / 18 | |
| if s > 0: | |
| z = (s - 1) / np.sqrt(var_s) | |
| elif s < 0: | |
| z = (s + 1) / np.sqrt(var_s) | |
| else: | |
| z = 0 | |
| p_value = 2 * (1 - stats.norm.cdf(abs(z))) | |
| return z, p_value | |
| mk_stat, mk_p_value = mann_kendall_test(data) | |
| return { | |
| 'slope': slope, | |
| 'intercept': intercept, | |
| 'r_squared': r_value ** 2, | |
| 'p_value': p_value, | |
| 'mann_kendall_stat': mk_stat, | |
| 'mann_kendall_p_value': mk_p_value, | |
| 'trend_strength': abs(slope) * r_value | |
| } | |
| class ClusteringAnalyzer: | |
| def __init__(self): | |
| self.scaler = StandardScaler() | |
| self.cluster_models = {} | |
| def perform_kmeans_clustering(self, data: np.ndarray, n_clusters: int = 3) -> Dict[str, Any]: | |
| data_scaled = self.scaler.fit_transform(data) | |
| kmeans = KMeans(n_clusters=n_clusters, random_state=42, n_init=10) | |
| cluster_labels = kmeans.fit_predict(data_scaled) | |
| from sklearn.metrics import silhouette_score | |
| silhouette_avg = silhouette_score(data_scaled, cluster_labels) | |
| self.cluster_models['kmeans'] = kmeans | |
| return { | |
| 'labels': cluster_labels, | |
| 'centers': kmeans.cluster_centers_, | |
| 'inertia': kmeans.inertia_, | |
| 'silhouette_score': silhouette_avg, | |
| 'n_clusters': n_clusters | |
| } | |
| def perform_hierarchical_clustering(self, data: np.ndarray, | |
| n_clusters: int = 3) -> Dict[str, Any]: | |
| data_scaled = self.scaler.fit_transform(data) | |
| hierarchical = AgglomerativeClustering(n_clusters=n_clusters) | |
| cluster_labels = hierarchical.fit_predict(data_scaled) | |
| linkage_matrix = linkage(data_scaled, method='ward') | |
| return { | |
| 'labels': cluster_labels, | |
| 'linkage_matrix': linkage_matrix, | |
| 'n_clusters': n_clusters | |
| } | |
| def perform_dbscan_clustering(self, data: np.ndarray, | |
| eps: float = 0.5, min_samples: int = 5) -> Dict[str, Any]: | |
| data_scaled = self.scaler.fit_transform(data) | |
| dbscan = DBSCAN(eps=eps, min_samples=min_samples) | |
| cluster_labels = dbscan.fit_predict(data_scaled) | |
| n_clusters = len(set(cluster_labels)) - (1 if -1 in cluster_labels else 0) | |
| n_noise = list(cluster_labels).count(-1) | |
| return { | |
| 'labels': cluster_labels, | |
| 'n_clusters': n_clusters, | |
| 'n_noise': n_noise, | |
| 'eps': eps, | |
| 'min_samples': min_samples | |
| } | |
| def find_optimal_clusters(self, data: np.ndarray, max_clusters: int = 10) -> Dict[str, Any]: | |
| data_scaled = self.scaler.fit_transform(data) | |
| inertias = [] | |
| silhouette_scores = [] | |
| k_range = range(2, max_clusters + 1) | |
| for k in k_range: | |
| kmeans = KMeans(n_clusters=k, random_state=42, n_init=10) | |
| cluster_labels = kmeans.fit_predict(data_scaled) | |
| inertias.append(kmeans.inertia_) | |
| if k > 1: | |
| from sklearn.metrics import silhouette_score | |
| silhouette_scores.append(silhouette_score(data_scaled, cluster_labels)) | |
| else: | |
| silhouette_scores.append(0) | |
| if len(inertias) > 1: | |
| second_derivative = np.diff(inertias, 2) | |
| elbow_k = k_range[np.argmax(second_derivative) + 2] | |
| else: | |
| elbow_k = 2 | |
| best_silhouette_k = k_range[np.argmax(silhouette_scores)] | |
| return { | |
| 'k_range': list(k_range), | |
| 'inertias': inertias, | |
| 'silhouette_scores': silhouette_scores, | |
| 'elbow_k': elbow_k, | |
| 'best_silhouette_k': best_silhouette_k | |
| } | |
| class DimensionalityReducer: | |
| def __init__(self): | |
| self.scaler = StandardScaler() | |
| self.reduction_models = {} | |
| def perform_pca(self, data: np.ndarray, n_components: int = None) -> Dict[str, Any]: | |
| data_scaled = self.scaler.fit_transform(data) | |
| pca = PCA(n_components=n_components) | |
| data_pca = pca.fit_transform(data_scaled) | |
| self.reduction_models['pca'] = pca | |
| return { | |
| 'transformed_data': data_pca, | |
| 'explained_variance_ratio': pca.explained_variance_ratio_, | |
| 'cumulative_variance': np.cumsum(pca.explained_variance_ratio_), | |
| 'components': pca.components_, | |
| 'n_components': pca.n_components_ | |
| } | |
| def perform_tsne(self, data: np.ndarray, n_components: int = 2, | |
| perplexity: float = 30.0) -> Dict[str, Any]: | |
| data_scaled = self.scaler.fit_transform(data) | |
| tsne = TSNE(n_components=n_components, perplexity=perplexity, | |
| random_state=42, verbose=0) | |
| data_tsne = tsne.fit_transform(data_scaled) | |
| self.reduction_models['tsne'] = tsne | |
| return { | |
| 'transformed_data': data_tsne, | |
| 'perplexity': perplexity, | |
| 'n_components': n_components | |
| } | |
| def perform_umap(self, data: np.ndarray, n_components: int = 2, | |
| n_neighbors: int = 15) -> Dict[str, Any]: | |
| try: | |
| import umap | |
| except ImportError: | |
| print("UMAP not available, using PCA instead") | |
| return self.perform_pca(data, n_components) | |
| data_scaled = self.scaler.fit_transform(data) | |
| umap_reducer = umap.UMAP(n_components=n_components, n_neighbors=n_neighbors, | |
| random_state=42) | |
| data_umap = umap_reducer.fit_transform(data_scaled) | |
| self.reduction_models['umap'] = umap_reducer | |
| return { | |
| 'transformed_data': data_umap, | |
| 'n_neighbors': n_neighbors, | |
| 'n_components': n_components | |
| } | |
| class PredictionModeler: | |
| def __init__(self): | |
| self.models = {} | |
| self.scaler = StandardScaler() | |
| def prepare_data(self, X: np.ndarray, y: np.ndarray, | |
| test_size: float = 0.2) -> Dict[str, np.ndarray]: | |
| X_scaled = self.scaler.fit_transform(X) | |
| X_train, X_test, y_train, y_test = train_test_split( | |
| X_scaled, y, test_size=test_size, random_state=42 | |
| ) | |
| return { | |
| 'X_train': X_train, | |
| 'X_test': X_test, | |
| 'y_train': y_train, | |
| 'y_test': y_test | |
| } | |
| def train_linear_models(self, X: np.ndarray, y: np.ndarray) -> Dict[str, Dict[str, float]]: | |
| data = self.prepare_data(X, y) | |
| models = { | |
| 'linear': LinearRegression(), | |
| 'ridge': Ridge(alpha=1.0), | |
| 'lasso': Lasso(alpha=0.1) | |
| } | |
| results = {} | |
| for name, model in models.items(): | |
| model.fit(data['X_train'], data['y_train']) | |
| y_pred = model.predict(data['X_test']) | |
| mse = mean_squared_error(data['y_test'], y_pred) | |
| rmse = np.sqrt(mse) | |
| mae = mean_absolute_error(data['y_test'], y_pred) | |
| r2 = r2_score(data['y_test'], y_pred) | |
| results[name] = { | |
| 'mse': mse, | |
| 'rmse': rmse, | |
| 'mae': mae, | |
| 'r2': r2 | |
| } | |
| self.models[name] = model | |
| return results | |
| def train_ensemble_models(self, X: np.ndarray, y: np.ndarray) -> Dict[str, Dict[str, float]]: | |
| data = self.prepare_data(X, y) | |
| models = { | |
| 'random_forest': RandomForestRegressor(n_estimators=100, random_state=42), | |
| 'gradient_boosting': GradientBoostingRegressor(n_estimators=100, random_state=42) | |
| } | |
| results = {} | |
| for name, model in models.items(): | |
| model.fit(data['X_train'], data['y_train']) | |
| y_pred = model.predict(data['X_test']) | |
| mse = mean_squared_error(data['y_test'], y_pred) | |
| rmse = np.sqrt(mse) | |
| mae = mean_absolute_error(data['y_test'], y_pred) | |
| r2 = r2_score(data['y_test'], y_pred) | |
| results[name] = { | |
| 'mse': mse, | |
| 'rmse': rmse, | |
| 'mae': mae, | |
| 'r2': r2 | |
| } | |
| if hasattr(model, 'feature_importances_'): | |
| feature_importance = dict(zip( | |
| range(len(model.feature_importances_)), | |
| model.feature_importances_ | |
| )) | |
| results[name]['feature_importance'] = feature_importance | |
| self.models[name] = model | |
| return results | |
| class AnomalyDetector: | |
| def __init__(self): | |
| self.detectors = {} | |
| def detect_outliers_iqr(self, data: np.ndarray, factor: float = 1.5) -> List[int]: | |
| Q1 = np.percentile(data, 25) | |
| Q3 = np.percentile(data, 75) | |
| IQR = Q3 - Q1 | |
| lower_bound = Q1 - factor * IQR | |
| upper_bound = Q3 + factor * IQR | |
| outliers = np.where((data < lower_bound) | (data > upper_bound))[0] | |
| return outliers.tolist() | |
| def detect_outliers_zscore(self, data: np.ndarray, threshold: float = 3.0) -> List[int]: | |
| z_scores = np.abs(stats.zscore(data)) | |
| outliers = np.where(z_scores > threshold)[0] | |
| return outliers.tolist() | |
| def detect_outliers_lof(self, data: np.ndarray, n_neighbors: int = 20) -> List[int]: | |
| if data.ndim == 1: | |
| data = data.reshape(-1, 1) | |
| lof = LocalOutlierFactor(n_neighbors=n_neighbors, contamination=0.1) | |
| outlier_labels = lof.fit_predict(data) | |
| outliers = np.where(outlier_labels == -1)[0] | |
| return outliers.tolist() | |
| def detect_outliers_svm(self, data: np.ndarray, nu: float = 0.1) -> List[int]: | |
| if data.ndim == 1: | |
| data = data.reshape(-1, 1) | |
| svm = OneClassSVM(nu=nu) | |
| outlier_labels = svm.fit_predict(data) | |
| outliers = np.where(outlier_labels == -1)[0] | |
| return outliers.tolist() | |
| def comprehensive_anomaly_detection(self, data: np.ndarray) -> Dict[str, List[int]]: | |
| results = {} | |
| results['iqr'] = self.detect_outliers_iqr(data) | |
| results['zscore'] = self.detect_outliers_zscore(data) | |
| results['lof'] = self.detect_outliers_lof(data) | |
| results['svm'] = self.detect_outliers_svm(data) | |
| all_outliers = [] | |
| for method, outliers in results.items(): | |
| all_outliers.extend(outliers) | |
| outlier_counts = {} | |
| for outlier in all_outliers: | |
| outlier_counts[outlier] = outlier_counts.get(outlier, 0) + 1 | |
| consensus_outliers = [outlier for outlier, count in outlier_counts.items() if count >= 2] | |
| results['consensus'] = consensus_outliers | |
| return results | |
| class ComprehensiveAnalyzer: | |
| def __init__(self): | |
| self.bayesian = BayesianAnalyzer() | |
| self.timeseries = TimeSeriesAnalyzer() | |
| self.clustering = ClusteringAnalyzer() | |
| self.dimensionality = DimensionalityReducer() | |
| self.prediction = PredictionModeler() | |
| self.anomaly = AnomalyDetector() | |
| def analyze_memory_performance(self, data: pd.DataFrame) -> StatisticalResults: | |
| results = StatisticalResults() | |
| numeric_data = data.select_dtypes(include=[np.number]) | |
| if len(numeric_data.columns) == 0: | |
| print("No numerical data found for analysis") | |
| return results | |
| results.mean = numeric_data.mean().mean() | |
| results.median = numeric_data.median().median() | |
| results.std = numeric_data.std().mean() | |
| results.skewness = numeric_data.skew().mean() | |
| results.kurtosis = numeric_data.kurtosis().mean() | |
| correlation_matrix = numeric_data.corr() | |
| results.correlation_matrix = correlation_matrix.values | |
| significant_correlations = [] | |
| for i in range(len(correlation_matrix.columns)): | |
| for j in range(i+1, len(correlation_matrix.columns)): | |
| corr_value = correlation_matrix.iloc[i, j] | |
| if abs(corr_value) > 0.5: | |
| significant_correlations.append(( | |
| correlation_matrix.columns[i], | |
| correlation_matrix.columns[j], | |
| corr_value | |
| )) | |
| results.significant_correlations = significant_correlations | |
| if len(numeric_data) > 10: | |
| optimal_clusters = self.clustering.find_optimal_clusters(numeric_data.values) | |
| n_clusters = optimal_clusters['best_silhouette_k'] | |
| clustering_results = self.clustering.perform_kmeans_clustering( | |
| numeric_data.values, n_clusters | |
| ) | |
| results.cluster_labels = clustering_results['labels'] | |
| results.silhouette_score = clustering_results['silhouette_score'] | |
| results.inertia = clustering_results['inertia'] | |
| if len(numeric_data.columns) > 2: | |
| pca_results = self.dimensionality.perform_pca(numeric_data.values) | |
| results.explained_variance_ratio = pca_results['explained_variance_ratio'] | |
| results.cumulative_variance = pca_results['cumulative_variance'] | |
| if len(numeric_data) > 20: | |
| ts_data = numeric_data.iloc[:, 0].values | |
| trend_results = self.timeseries.detect_trend(ts_data) | |
| results.trend_coefficient = trend_results['slope'] | |
| autocorr = self.timeseries.calculate_autocorrelation(ts_data) | |
| results.autocorrelation_lags = autocorr | |
| if len(numeric_data.columns) > 1: | |
| X = numeric_data.iloc[:, :-1].values | |
| y = numeric_data.iloc[:, -1].values | |
| linear_results = self.prediction.train_linear_models(X, y) | |
| results.model_performance.update(linear_results) | |
| ensemble_results = self.prediction.train_ensemble_models(X, y) | |
| results.model_performance.update(ensemble_results) | |
| for model_name, model_results in ensemble_results.items(): | |
| if 'feature_importance' in model_results: | |
| results.feature_importance.update(model_results['feature_importance']) | |
| if len(numeric_data) > 10: | |
| anomaly_data = numeric_data.iloc[:, 0].values | |
| anomaly_results = self.anomaly.comprehensive_anomaly_detection(anomaly_data) | |
| results.anomaly_indices = anomaly_results['consensus'] | |
| results.anomaly_scores = np.zeros(len(anomaly_data)) | |
| for idx in results.anomaly_indices: | |
| results.anomaly_scores[idx] = 1.0 | |
| return results | |
| def generate_report(self, results: StatisticalResults) -> str: | |
| report = [] | |
| report.append("📊 COMPREHENSIVE STATISTICAL ANALYSIS REPORT") | |
| report.append("=" * 60) | |
| report.append("\n📈 DESCRIPTIVE STATISTICS") | |
| report.append("-" * 30) | |
| report.append(f"Mean: {results.mean:.4f}") | |
| report.append(f"Median: {results.median:.4f}") | |
| report.append(f"Standard Deviation: {results.std:.4f}") | |
| report.append(f"Skewness: {results.skewness:.4f}") | |
| report.append(f"Kurtosis: {results.kurtosis:.4f}") | |
| report.append("\n🔗 CORRELATION ANALYSIS") | |
| report.append("-" * 30) | |
| report.append(f"Number of significant correlations: {len(results.significant_correlations)}") | |
| for corr in results.significant_correlations[:5]: | |
| report.append(f" {corr[0]} ↔ {corr[1]}: {corr[2]:.3f}") | |
| if len(results.cluster_labels) > 0: | |
| report.append("\n🎯 CLUSTERING ANALYSIS") | |
| report.append("-" * 30) | |
| report.append(f"Number of clusters: {len(set(results.cluster_labels))}") | |
| report.append(f"Silhouette Score: {results.silhouette_score:.4f}") | |
| report.append(f"Inertia: {results.inertia:.4f}") | |
| if len(results.explained_variance_ratio) > 0: | |
| report.append("\n📉 DIMENSIONALITY REDUCTION") | |
| report.append("-" * 30) | |
| report.append(f"First 3 components explain: {np.sum(results.explained_variance_ratio[:3]):.1%} of variance") | |
| report.append(f"Components needed for 95% variance: {np.argmax(results.cumulative_variance >= 0.95) + 1}") | |
| if results.trend_coefficient != 0: | |
| report.append("\n📈 TIME SERIES ANALYSIS") | |
| report.append("-" * 30) | |
| report.append(f"Trend coefficient: {results.trend_coefficient:.4f}") | |
| report.append(f"Number of autocorrelation lags: {len(results.autocorrelation_lags)}") | |
| if results.model_performance: | |
| report.append("\n🤖 PREDICTION MODELS") | |
| report.append("-" * 30) | |
| for model_name, metrics in results.model_performance.items(): | |
| report.append(f"{model_name.upper()}:") | |
| report.append(f" R² Score: {metrics.get('r2', 0):.4f}") | |
| report.append(f" RMSE: {metrics.get('rmse', 0):.4f}") | |
| if len(results.anomaly_indices) > 0: | |
| report.append("\n🚨 ANOMALY DETECTION") | |
| report.append("-" * 30) | |
| report.append(f"Number of anomalies detected: {len(results.anomaly_indices)}") | |
| report.append(f"Anomaly rate: {len(results.anomaly_indices)/len(results.anomaly_scores):.1%}") | |
| return "\n".join(report) | |
| def run_comprehensive_analysis(): | |
| print("📊 Advanced Statistical Analysis Demo") | |
| print("=" * 50) | |
| np.random.seed(42) | |
| n_samples = 1000 | |
| data = { | |
| 'memory_usage': np.random.gamma(2, 2, n_samples), | |
| 'access_time': np.random.exponential(0.1, n_samples), | |
| 'hit_rate': np.random.beta(2, 1, n_samples), | |
| 'throughput': np.random.normal(1000, 200, n_samples), | |
| 'latency': np.random.lognormal(0, 0.5, n_samples), | |
| 'cache_size': np.random.uniform(100, 10000, n_samples), | |
| 'compression_ratio': np.random.beta(3, 2, n_samples), | |
| 'energy_consumption': np.random.gamma(1.5, 1, n_samples) | |
| } | |
| data['memory_usage'] += data['cache_size'] * 0.1 | |
| data['access_time'] += data['latency'] * 0.3 | |
| data['hit_rate'] -= data['latency'] * 0.2 | |
| anomaly_indices = np.random.choice(n_samples, size=20, replace=False) | |
| data['memory_usage'][anomaly_indices] *= 3 | |
| data['access_time'][anomaly_indices] *= 5 | |
| df = pd.DataFrame(data) | |
| analyzer = ComprehensiveAnalyzer() | |
| results = analyzer.analyze_memory_performance(df) | |
| report = analyzer.generate_report(results) | |
| print(report) | |
| with open('statistical_analysis_results.json', 'w') as f: | |
| results_dict = { | |
| 'mean': results.mean, | |
| 'median': results.median, | |
| 'std': results.std, | |
| 'skewness': results.skewness, | |
| 'kurtosis': results.kurtosis, | |
| 'significant_correlations': results.significant_correlations, | |
| 'silhouette_score': results.silhouette_score, | |
| 'inertia': results.inertia, | |
| 'trend_coefficient': results.trend_coefficient, | |
| 'model_performance': results.model_performance, | |
| 'feature_importance': results.feature_importance, | |
| 'anomaly_count': len(results.anomaly_indices) | |
| } | |
| json.dump(results_dict, f, indent=2) | |
| print(f"\n💾 Results saved to 'statistical_analysis_results.json'") | |
| print("✅ Comprehensive statistical analysis completed!") | |
| if __name__ == "__main__": | |
| run_comprehensive_analysis() | |
Xet Storage Details
- Size:
- 28.2 kB
- Xet hash:
- 5ce4ccb00cd0f3417012fded0d165229e4365ea52faacca4dd933631e548d9c9
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.