tahamajs's picture
download
raw
28.2 kB
#!/usr/bin/env python3
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
import seaborn as sns
from scipy import stats
from scipy.cluster.hierarchy import dendrogram, linkage
from sklearn.cluster import KMeans, DBSCAN, AgglomerativeClustering
from sklearn.decomposition import PCA, t-SNE, UMAP
from sklearn.manifold import TSNE
from sklearn.preprocessing import StandardScaler
from sklearn.model_selection import train_test_split
from sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor
from sklearn.linear_model import LinearRegression, Ridge, Lasso
from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error
from sklearn.neighbors import LocalOutlierFactor
from sklearn.svm import OneClassSVM
import warnings
warnings.filterwarnings('ignore')
from typing import Dict, List, Tuple, Any, Optional, Union
import json
import pickle
from dataclasses import dataclass
from datetime import datetime, timedelta
import plotly.graph_objects as go
import plotly.express as px
from plotly.subplots import make_subplots
@dataclass
class StatisticalResults:
mean: float = 0.0
median: float = 0.0
std: float = 0.0
skewness: float = 0.0
kurtosis: float = 0.0
correlation_matrix: np.ndarray = None
significant_correlations: List[Tuple[str, str, float]] = None
cluster_labels: np.ndarray = None
silhouette_score: float = 0.0
inertia: float = 0.0
explained_variance_ratio: np.ndarray = None
cumulative_variance: np.ndarray = None
trend_coefficient: float = 0.0
seasonality_strength: float = 0.0
autocorrelation_lags: Dict[int, float] = None
model_performance: Dict[str, float] = None
feature_importance: Dict[str, float] = None
anomaly_scores: np.ndarray = None
anomaly_indices: List[int] = None
def __post_init__(self):
if self.correlation_matrix is None:
self.correlation_matrix = np.array([])
if self.significant_correlations is None:
self.significant_correlations = []
if self.cluster_labels is None:
self.cluster_labels = np.array([])
if self.explained_variance_ratio is None:
self.explained_variance_ratio = np.array([])
if self.cumulative_variance is None:
self.cumulative_variance = np.array([])
if self.autocorrelation_lags is None:
self.autocorrelation_lags = {}
if self.model_performance is None:
self.model_performance = {}
if self.feature_importance is None:
self.feature_importance = {}
if self.anomaly_scores is None:
self.anomaly_scores = np.array([])
if self.anomaly_indices is None:
self.anomaly_indices = []
class BayesianAnalyzer:
def __init__(self):
self.prior_distributions = {}
self.posterior_samples = {}
def set_prior(self, parameter: str, distribution_type: str, **params):
if distribution_type == "normal":
self.prior_distributions[parameter] = {
'type': 'normal',
'mean': params.get('mean', 0),
'std': params.get('std', 1)
}
elif distribution_type == "beta":
self.prior_distributions[parameter] = {
'type': 'beta',
'alpha': params.get('alpha', 1),
'beta': params.get('beta', 1)
}
elif distribution_type == "gamma":
self.prior_distributions[parameter] = {
'type': 'gamma',
'shape': params.get('shape', 1),
'scale': params.get('scale', 1)
}
def bayesian_inference(self, data: np.ndarray, parameter: str,
n_samples: int = 10000) -> np.ndarray:
if parameter not in self.prior_distributions:
raise ValueError(f"No prior set for parameter: {parameter}")
prior = self.prior_distributions[parameter]
samples = []
current_value = np.mean(data)
for _ in range(n_samples):
if prior['type'] == 'normal':
proposal = current_value + np.random.normal(0, 0.1)
prior_prob_current = stats.norm.pdf(current_value,
prior['mean'], prior['std'])
prior_prob_proposal = stats.norm.pdf(proposal,
prior['mean'], prior['std'])
else:
proposal = current_value + np.random.normal(0, 0.1)
prior_prob_current = 1.0
prior_prob_proposal = 1.0
likelihood_current = np.prod(stats.norm.pdf(data, current_value, np.std(data)))
likelihood_proposal = np.prod(stats.norm.pdf(data, proposal, np.std(data)))
ratio = (likelihood_proposal * prior_prob_proposal) / \
(likelihood_current * prior_prob_current)
if np.random.random() < min(1, ratio):
current_value = proposal
samples.append(current_value)
self.posterior_samples[parameter] = np.array(samples)
return np.array(samples)
def credible_interval(self, parameter: str, alpha: float = 0.05) -> Tuple[float, float]:
if parameter not in self.posterior_samples:
raise ValueError(f"No posterior samples for parameter: {parameter}")
samples = self.posterior_samples[parameter]
lower = np.percentile(samples, (alpha/2) * 100)
upper = np.percentile(samples, (1 - alpha/2) * 100)
return lower, upper
class TimeSeriesAnalyzer:
def __init__(self):
self.trend_model = None
self.seasonal_model = None
def decompose_timeseries(self, data: np.ndarray, period: int = None) -> Dict[str, np.ndarray]:
n = len(data)
if period is None:
period = max(2, n // 10)
trend = np.convolve(data, np.ones(period)/period, mode='same')
detrended = data - trend
seasonal = np.zeros_like(data)
if period > 1:
for i in range(period):
seasonal[i::period] = np.mean(detrended[i::period])
residual = data - trend - seasonal
return {
'trend': trend,
'seasonal': seasonal,
'residual': residual,
'original': data
}
def calculate_autocorrelation(self, data: np.ndarray, max_lags: int = 20) -> Dict[int, float]:
autocorr = {}
n = len(data)
for lag in range(max_lags + 1):
if lag == 0:
autocorr[lag] = 1.0
else:
numerator = np.sum((data[lag:] - np.mean(data)) *
(data[:-lag] - np.mean(data)))
denominator = np.sum((data - np.mean(data)) ** 2)
if denominator != 0:
autocorr[lag] = numerator / denominator
else:
autocorr[lag] = 0.0
return autocorr
def detect_trend(self, data: np.ndarray) -> Dict[str, float]:
n = len(data)
x = np.arange(n)
slope, intercept, r_value, p_value, std_err = stats.linregress(x, data)
def mann_kendall_test(y):
n = len(y)
s = 0
for i in range(n-1):
for j in range(i+1, n):
s += np.sign(y[j] - y[i])
var_s = n * (n - 1) * (2 * n + 5) / 18
if s > 0:
z = (s - 1) / np.sqrt(var_s)
elif s < 0:
z = (s + 1) / np.sqrt(var_s)
else:
z = 0
p_value = 2 * (1 - stats.norm.cdf(abs(z)))
return z, p_value
mk_stat, mk_p_value = mann_kendall_test(data)
return {
'slope': slope,
'intercept': intercept,
'r_squared': r_value ** 2,
'p_value': p_value,
'mann_kendall_stat': mk_stat,
'mann_kendall_p_value': mk_p_value,
'trend_strength': abs(slope) * r_value
}
class ClusteringAnalyzer:
def __init__(self):
self.scaler = StandardScaler()
self.cluster_models = {}
def perform_kmeans_clustering(self, data: np.ndarray, n_clusters: int = 3) -> Dict[str, Any]:
data_scaled = self.scaler.fit_transform(data)
kmeans = KMeans(n_clusters=n_clusters, random_state=42, n_init=10)
cluster_labels = kmeans.fit_predict(data_scaled)
from sklearn.metrics import silhouette_score
silhouette_avg = silhouette_score(data_scaled, cluster_labels)
self.cluster_models['kmeans'] = kmeans
return {
'labels': cluster_labels,
'centers': kmeans.cluster_centers_,
'inertia': kmeans.inertia_,
'silhouette_score': silhouette_avg,
'n_clusters': n_clusters
}
def perform_hierarchical_clustering(self, data: np.ndarray,
n_clusters: int = 3) -> Dict[str, Any]:
data_scaled = self.scaler.fit_transform(data)
hierarchical = AgglomerativeClustering(n_clusters=n_clusters)
cluster_labels = hierarchical.fit_predict(data_scaled)
linkage_matrix = linkage(data_scaled, method='ward')
return {
'labels': cluster_labels,
'linkage_matrix': linkage_matrix,
'n_clusters': n_clusters
}
def perform_dbscan_clustering(self, data: np.ndarray,
eps: float = 0.5, min_samples: int = 5) -> Dict[str, Any]:
data_scaled = self.scaler.fit_transform(data)
dbscan = DBSCAN(eps=eps, min_samples=min_samples)
cluster_labels = dbscan.fit_predict(data_scaled)
n_clusters = len(set(cluster_labels)) - (1 if -1 in cluster_labels else 0)
n_noise = list(cluster_labels).count(-1)
return {
'labels': cluster_labels,
'n_clusters': n_clusters,
'n_noise': n_noise,
'eps': eps,
'min_samples': min_samples
}
def find_optimal_clusters(self, data: np.ndarray, max_clusters: int = 10) -> Dict[str, Any]:
data_scaled = self.scaler.fit_transform(data)
inertias = []
silhouette_scores = []
k_range = range(2, max_clusters + 1)
for k in k_range:
kmeans = KMeans(n_clusters=k, random_state=42, n_init=10)
cluster_labels = kmeans.fit_predict(data_scaled)
inertias.append(kmeans.inertia_)
if k > 1:
from sklearn.metrics import silhouette_score
silhouette_scores.append(silhouette_score(data_scaled, cluster_labels))
else:
silhouette_scores.append(0)
if len(inertias) > 1:
second_derivative = np.diff(inertias, 2)
elbow_k = k_range[np.argmax(second_derivative) + 2]
else:
elbow_k = 2
best_silhouette_k = k_range[np.argmax(silhouette_scores)]
return {
'k_range': list(k_range),
'inertias': inertias,
'silhouette_scores': silhouette_scores,
'elbow_k': elbow_k,
'best_silhouette_k': best_silhouette_k
}
class DimensionalityReducer:
def __init__(self):
self.scaler = StandardScaler()
self.reduction_models = {}
def perform_pca(self, data: np.ndarray, n_components: int = None) -> Dict[str, Any]:
data_scaled = self.scaler.fit_transform(data)
pca = PCA(n_components=n_components)
data_pca = pca.fit_transform(data_scaled)
self.reduction_models['pca'] = pca
return {
'transformed_data': data_pca,
'explained_variance_ratio': pca.explained_variance_ratio_,
'cumulative_variance': np.cumsum(pca.explained_variance_ratio_),
'components': pca.components_,
'n_components': pca.n_components_
}
def perform_tsne(self, data: np.ndarray, n_components: int = 2,
perplexity: float = 30.0) -> Dict[str, Any]:
data_scaled = self.scaler.fit_transform(data)
tsne = TSNE(n_components=n_components, perplexity=perplexity,
random_state=42, verbose=0)
data_tsne = tsne.fit_transform(data_scaled)
self.reduction_models['tsne'] = tsne
return {
'transformed_data': data_tsne,
'perplexity': perplexity,
'n_components': n_components
}
def perform_umap(self, data: np.ndarray, n_components: int = 2,
n_neighbors: int = 15) -> Dict[str, Any]:
try:
import umap
except ImportError:
print("UMAP not available, using PCA instead")
return self.perform_pca(data, n_components)
data_scaled = self.scaler.fit_transform(data)
umap_reducer = umap.UMAP(n_components=n_components, n_neighbors=n_neighbors,
random_state=42)
data_umap = umap_reducer.fit_transform(data_scaled)
self.reduction_models['umap'] = umap_reducer
return {
'transformed_data': data_umap,
'n_neighbors': n_neighbors,
'n_components': n_components
}
class PredictionModeler:
def __init__(self):
self.models = {}
self.scaler = StandardScaler()
def prepare_data(self, X: np.ndarray, y: np.ndarray,
test_size: float = 0.2) -> Dict[str, np.ndarray]:
X_scaled = self.scaler.fit_transform(X)
X_train, X_test, y_train, y_test = train_test_split(
X_scaled, y, test_size=test_size, random_state=42
)
return {
'X_train': X_train,
'X_test': X_test,
'y_train': y_train,
'y_test': y_test
}
def train_linear_models(self, X: np.ndarray, y: np.ndarray) -> Dict[str, Dict[str, float]]:
data = self.prepare_data(X, y)
models = {
'linear': LinearRegression(),
'ridge': Ridge(alpha=1.0),
'lasso': Lasso(alpha=0.1)
}
results = {}
for name, model in models.items():
model.fit(data['X_train'], data['y_train'])
y_pred = model.predict(data['X_test'])
mse = mean_squared_error(data['y_test'], y_pred)
rmse = np.sqrt(mse)
mae = mean_absolute_error(data['y_test'], y_pred)
r2 = r2_score(data['y_test'], y_pred)
results[name] = {
'mse': mse,
'rmse': rmse,
'mae': mae,
'r2': r2
}
self.models[name] = model
return results
def train_ensemble_models(self, X: np.ndarray, y: np.ndarray) -> Dict[str, Dict[str, float]]:
data = self.prepare_data(X, y)
models = {
'random_forest': RandomForestRegressor(n_estimators=100, random_state=42),
'gradient_boosting': GradientBoostingRegressor(n_estimators=100, random_state=42)
}
results = {}
for name, model in models.items():
model.fit(data['X_train'], data['y_train'])
y_pred = model.predict(data['X_test'])
mse = mean_squared_error(data['y_test'], y_pred)
rmse = np.sqrt(mse)
mae = mean_absolute_error(data['y_test'], y_pred)
r2 = r2_score(data['y_test'], y_pred)
results[name] = {
'mse': mse,
'rmse': rmse,
'mae': mae,
'r2': r2
}
if hasattr(model, 'feature_importances_'):
feature_importance = dict(zip(
range(len(model.feature_importances_)),
model.feature_importances_
))
results[name]['feature_importance'] = feature_importance
self.models[name] = model
return results
class AnomalyDetector:
def __init__(self):
self.detectors = {}
def detect_outliers_iqr(self, data: np.ndarray, factor: float = 1.5) -> List[int]:
Q1 = np.percentile(data, 25)
Q3 = np.percentile(data, 75)
IQR = Q3 - Q1
lower_bound = Q1 - factor * IQR
upper_bound = Q3 + factor * IQR
outliers = np.where((data < lower_bound) | (data > upper_bound))[0]
return outliers.tolist()
def detect_outliers_zscore(self, data: np.ndarray, threshold: float = 3.0) -> List[int]:
z_scores = np.abs(stats.zscore(data))
outliers = np.where(z_scores > threshold)[0]
return outliers.tolist()
def detect_outliers_lof(self, data: np.ndarray, n_neighbors: int = 20) -> List[int]:
if data.ndim == 1:
data = data.reshape(-1, 1)
lof = LocalOutlierFactor(n_neighbors=n_neighbors, contamination=0.1)
outlier_labels = lof.fit_predict(data)
outliers = np.where(outlier_labels == -1)[0]
return outliers.tolist()
def detect_outliers_svm(self, data: np.ndarray, nu: float = 0.1) -> List[int]:
if data.ndim == 1:
data = data.reshape(-1, 1)
svm = OneClassSVM(nu=nu)
outlier_labels = svm.fit_predict(data)
outliers = np.where(outlier_labels == -1)[0]
return outliers.tolist()
def comprehensive_anomaly_detection(self, data: np.ndarray) -> Dict[str, List[int]]:
results = {}
results['iqr'] = self.detect_outliers_iqr(data)
results['zscore'] = self.detect_outliers_zscore(data)
results['lof'] = self.detect_outliers_lof(data)
results['svm'] = self.detect_outliers_svm(data)
all_outliers = []
for method, outliers in results.items():
all_outliers.extend(outliers)
outlier_counts = {}
for outlier in all_outliers:
outlier_counts[outlier] = outlier_counts.get(outlier, 0) + 1
consensus_outliers = [outlier for outlier, count in outlier_counts.items() if count >= 2]
results['consensus'] = consensus_outliers
return results
class ComprehensiveAnalyzer:
def __init__(self):
self.bayesian = BayesianAnalyzer()
self.timeseries = TimeSeriesAnalyzer()
self.clustering = ClusteringAnalyzer()
self.dimensionality = DimensionalityReducer()
self.prediction = PredictionModeler()
self.anomaly = AnomalyDetector()
def analyze_memory_performance(self, data: pd.DataFrame) -> StatisticalResults:
results = StatisticalResults()
numeric_data = data.select_dtypes(include=[np.number])
if len(numeric_data.columns) == 0:
print("No numerical data found for analysis")
return results
results.mean = numeric_data.mean().mean()
results.median = numeric_data.median().median()
results.std = numeric_data.std().mean()
results.skewness = numeric_data.skew().mean()
results.kurtosis = numeric_data.kurtosis().mean()
correlation_matrix = numeric_data.corr()
results.correlation_matrix = correlation_matrix.values
significant_correlations = []
for i in range(len(correlation_matrix.columns)):
for j in range(i+1, len(correlation_matrix.columns)):
corr_value = correlation_matrix.iloc[i, j]
if abs(corr_value) > 0.5:
significant_correlations.append((
correlation_matrix.columns[i],
correlation_matrix.columns[j],
corr_value
))
results.significant_correlations = significant_correlations
if len(numeric_data) > 10:
optimal_clusters = self.clustering.find_optimal_clusters(numeric_data.values)
n_clusters = optimal_clusters['best_silhouette_k']
clustering_results = self.clustering.perform_kmeans_clustering(
numeric_data.values, n_clusters
)
results.cluster_labels = clustering_results['labels']
results.silhouette_score = clustering_results['silhouette_score']
results.inertia = clustering_results['inertia']
if len(numeric_data.columns) > 2:
pca_results = self.dimensionality.perform_pca(numeric_data.values)
results.explained_variance_ratio = pca_results['explained_variance_ratio']
results.cumulative_variance = pca_results['cumulative_variance']
if len(numeric_data) > 20:
ts_data = numeric_data.iloc[:, 0].values
trend_results = self.timeseries.detect_trend(ts_data)
results.trend_coefficient = trend_results['slope']
autocorr = self.timeseries.calculate_autocorrelation(ts_data)
results.autocorrelation_lags = autocorr
if len(numeric_data.columns) > 1:
X = numeric_data.iloc[:, :-1].values
y = numeric_data.iloc[:, -1].values
linear_results = self.prediction.train_linear_models(X, y)
results.model_performance.update(linear_results)
ensemble_results = self.prediction.train_ensemble_models(X, y)
results.model_performance.update(ensemble_results)
for model_name, model_results in ensemble_results.items():
if 'feature_importance' in model_results:
results.feature_importance.update(model_results['feature_importance'])
if len(numeric_data) > 10:
anomaly_data = numeric_data.iloc[:, 0].values
anomaly_results = self.anomaly.comprehensive_anomaly_detection(anomaly_data)
results.anomaly_indices = anomaly_results['consensus']
results.anomaly_scores = np.zeros(len(anomaly_data))
for idx in results.anomaly_indices:
results.anomaly_scores[idx] = 1.0
return results
def generate_report(self, results: StatisticalResults) -> str:
report = []
report.append("📊 COMPREHENSIVE STATISTICAL ANALYSIS REPORT")
report.append("=" * 60)
report.append("\n📈 DESCRIPTIVE STATISTICS")
report.append("-" * 30)
report.append(f"Mean: {results.mean:.4f}")
report.append(f"Median: {results.median:.4f}")
report.append(f"Standard Deviation: {results.std:.4f}")
report.append(f"Skewness: {results.skewness:.4f}")
report.append(f"Kurtosis: {results.kurtosis:.4f}")
report.append("\n🔗 CORRELATION ANALYSIS")
report.append("-" * 30)
report.append(f"Number of significant correlations: {len(results.significant_correlations)}")
for corr in results.significant_correlations[:5]:
report.append(f" {corr[0]} ↔ {corr[1]}: {corr[2]:.3f}")
if len(results.cluster_labels) > 0:
report.append("\n🎯 CLUSTERING ANALYSIS")
report.append("-" * 30)
report.append(f"Number of clusters: {len(set(results.cluster_labels))}")
report.append(f"Silhouette Score: {results.silhouette_score:.4f}")
report.append(f"Inertia: {results.inertia:.4f}")
if len(results.explained_variance_ratio) > 0:
report.append("\n📉 DIMENSIONALITY REDUCTION")
report.append("-" * 30)
report.append(f"First 3 components explain: {np.sum(results.explained_variance_ratio[:3]):.1%} of variance")
report.append(f"Components needed for 95% variance: {np.argmax(results.cumulative_variance >= 0.95) + 1}")
if results.trend_coefficient != 0:
report.append("\n📈 TIME SERIES ANALYSIS")
report.append("-" * 30)
report.append(f"Trend coefficient: {results.trend_coefficient:.4f}")
report.append(f"Number of autocorrelation lags: {len(results.autocorrelation_lags)}")
if results.model_performance:
report.append("\n🤖 PREDICTION MODELS")
report.append("-" * 30)
for model_name, metrics in results.model_performance.items():
report.append(f"{model_name.upper()}:")
report.append(f" R² Score: {metrics.get('r2', 0):.4f}")
report.append(f" RMSE: {metrics.get('rmse', 0):.4f}")
if len(results.anomaly_indices) > 0:
report.append("\n🚨 ANOMALY DETECTION")
report.append("-" * 30)
report.append(f"Number of anomalies detected: {len(results.anomaly_indices)}")
report.append(f"Anomaly rate: {len(results.anomaly_indices)/len(results.anomaly_scores):.1%}")
return "\n".join(report)
def run_comprehensive_analysis():
print("📊 Advanced Statistical Analysis Demo")
print("=" * 50)
np.random.seed(42)
n_samples = 1000
data = {
'memory_usage': np.random.gamma(2, 2, n_samples),
'access_time': np.random.exponential(0.1, n_samples),
'hit_rate': np.random.beta(2, 1, n_samples),
'throughput': np.random.normal(1000, 200, n_samples),
'latency': np.random.lognormal(0, 0.5, n_samples),
'cache_size': np.random.uniform(100, 10000, n_samples),
'compression_ratio': np.random.beta(3, 2, n_samples),
'energy_consumption': np.random.gamma(1.5, 1, n_samples)
}
data['memory_usage'] += data['cache_size'] * 0.1
data['access_time'] += data['latency'] * 0.3
data['hit_rate'] -= data['latency'] * 0.2
anomaly_indices = np.random.choice(n_samples, size=20, replace=False)
data['memory_usage'][anomaly_indices] *= 3
data['access_time'][anomaly_indices] *= 5
df = pd.DataFrame(data)
analyzer = ComprehensiveAnalyzer()
results = analyzer.analyze_memory_performance(df)
report = analyzer.generate_report(results)
print(report)
with open('statistical_analysis_results.json', 'w') as f:
results_dict = {
'mean': results.mean,
'median': results.median,
'std': results.std,
'skewness': results.skewness,
'kurtosis': results.kurtosis,
'significant_correlations': results.significant_correlations,
'silhouette_score': results.silhouette_score,
'inertia': results.inertia,
'trend_coefficient': results.trend_coefficient,
'model_performance': results.model_performance,
'feature_importance': results.feature_importance,
'anomaly_count': len(results.anomaly_indices)
}
json.dump(results_dict, f, indent=2)
print(f"\n💾 Results saved to 'statistical_analysis_results.json'")
print("✅ Comprehensive statistical analysis completed!")
if __name__ == "__main__":
run_comprehensive_analysis()

Xet Storage Details

Size:
28.2 kB
·
Xet hash:
5ce4ccb00cd0f3417012fded0d165229e4365ea52faacca4dd933631e548d9c9

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.