""" Evaluation framework to measure system performance on test dataset. Tracks success rate, retries, failure types, and latency. """ import json import time from typing import Any, Dict, List, Optional from datetime import datetime import sys import os # Add src to path sys.path.insert(0, os.path.join(os.path.dirname(__file__), '..', 'src')) from pipeline import Pipeline from runtime_simulator import validate_config_executable from test_dataset import get_test_dataset, get_real_prompts, get_edge_cases class EvaluationFramework: """Comprehensive evaluation of the system.""" def __init__(self, use_llm: bool = True): self.pipeline = Pipeline(use_llm=use_llm) self.results = [] self.summary = { "total_prompts": 0, "successful": 0, "failed": 0, "executable": 0, "total_retries": 0, "total_latency": 0.0, "by_category": {}, "failure_types": {}, "timestamp": datetime.now().isoformat(), } def evaluate_prompt(self, prompt: Dict[str, Any], max_retries: int = 1) -> Dict[str, Any]: """Evaluate a single prompt.""" prompt_id = prompt.get("id", "unknown") prompt_text = prompt.get("prompt", "") category = prompt.get("category", prompt.get("type", "unknown")) result = { "prompt_id": prompt_id, "category": category, "prompt_summary": prompt_text[:100], "success": False, "executable": False, "retries": 0, "latency": 0.0, "errors": [], "warnings": [], "config_size": 0, } start_time = time.time() # Try generation with retries for attempt in range(max_retries): result["retries"] = attempt + 1 try: config, exec_log = self.pipeline.generate(prompt_text) if not config: result["errors"].append("Empty config generated") continue # Check if executable is_executable, exec_report = validate_config_executable(config) result["success"] = True result["executable"] = is_executable result["config_size"] = len(json.dumps(config)) if not is_executable: result["warnings"].extend(exec_report.get("warnings", [])) result["errors"].extend(exec_report.get("errors", [])) # Store execution log result["execution_log"] = exec_log result["execution_report"] = exec_report break except Exception as e: error_msg = str(e) result["errors"].append(error_msg) # Categorize error error_type = self._categorize_error(error_msg) if error_type not in self.summary["failure_types"]: self.summary["failure_types"][error_type] = 0 self.summary["failure_types"][error_type] += 1 result["latency"] = time.time() - start_time return result def _categorize_error(self, error: str) -> str: """Categorize error type.""" error_lower = error.lower() if "json" in error_lower: return "json_error" elif "validation" in error_lower: return "validation_error" elif "field" in error_lower: return "field_error" elif "api" in error_lower: return "api_error" elif "database" in error_lower or "table" in error_lower: return "database_error" else: return "unknown_error" def run_evaluation(self, dataset_size: str = "full") -> Dict[str, Any]: """Run full evaluation on test dataset.""" if dataset_size == "full": prompts = get_real_prompts() + get_edge_cases() elif dataset_size == "real": prompts = get_real_prompts() elif dataset_size == "edge": prompts = get_edge_cases() else: prompts = get_real_prompts()[:int(dataset_size)] self.summary["total_prompts"] = len(prompts) print(f"\nšŸ“Š Running evaluation on {len(prompts)} prompts...") for i, prompt in enumerate(prompts): print(f" [{i+1}/{len(prompts)}] {prompt.get('name', prompt.get('id'))}", end=" ") result = self.evaluate_prompt(prompt) self.results.append(result) # Update summary if result["success"]: self.summary["successful"] += 1 print("āœ“") else: self.summary["failed"] += 1 print("āœ—") if result["executable"]: self.summary["executable"] += 1 self.summary["total_retries"] += result["retries"] self.summary["total_latency"] += result["latency"] # Track by category category = result["category"] if category not in self.summary["by_category"]: self.summary["by_category"][category] = {"success": 0, "total": 0} self.summary["by_category"][category]["total"] += 1 if result["success"]: self.summary["by_category"][category]["success"] += 1 # Calculate metrics self.summary["success_rate"] = (self.summary["successful"] / self.summary["total_prompts"]) * 100 if self.summary["total_prompts"] > 0 else 0 self.summary["executable_rate"] = (self.summary["executable"] / self.summary["total_prompts"]) * 100 if self.summary["total_prompts"] > 0 else 0 self.summary["avg_retries"] = self.summary["total_retries"] / self.summary["total_prompts"] if self.summary["total_prompts"] > 0 else 0 self.summary["avg_latency"] = self.summary["total_latency"] / self.summary["total_prompts"] if self.summary["total_prompts"] > 0 else 0 return self.get_report() def get_report(self) -> Dict[str, Any]: """Generate evaluation report.""" return { "summary": self.summary, "detailed_results": self.results, "cost_analysis": self._calculate_cost_analysis(), } def _calculate_cost_analysis(self) -> Dict[str, Any]: """Analyze cost vs quality tradeoff.""" if not self.results: return {} successful_configs = [r for r in self.results if r["success"]] if not successful_configs: return {"note": "No successful generations to analyze"} avg_config_size = sum(r["config_size"] for r in successful_configs) / len(successful_configs) avg_latency = sum(r["latency"] for r in successful_configs) / len(successful_configs) return { "avg_config_size_bytes": avg_config_size, "avg_generation_latency_seconds": round(avg_latency, 2), "estimated_api_calls_per_prompt": 4, # 4 stages "estimated_tokens_per_generation": int(avg_config_size / 4), # Rough estimate "quality_score": (self.summary["success_rate"] * 0.6) + (self.summary["executable_rate"] * 0.4), "efficiency_score": 100 - (avg_latency * 10), # Arbitrary scale "recommendation": self._get_recommendation(), } def _get_recommendation(self) -> str: """Get recommendation based on metrics.""" success_rate = self.summary.get("success_rate", 0) executable_rate = self.summary.get("executable_rate", 0) if success_rate >= 80 and executable_rate >= 75: return "Production-ready with monitoring" elif success_rate >= 60 and executable_rate >= 50: return "Ready for limited production use" elif success_rate >= 40: return "Needs refinement before production" else: return "Requires significant improvements" def print_report(self): """Print formatted report.""" print("\n" + "="*70) print("šŸ“Š EVALUATION REPORT") print("="*70) s = self.summary print(f"\nšŸ“ˆ SUMMARY METRICS:") print(f" Total Prompts Evaluated: {s['total_prompts']}") print(f" Successful Generations: {s['successful']}/{s['total_prompts']} ({s.get('success_rate', 0):.1f}%)") print(f" Executable Configs: {s['executable']}/{s['total_prompts']} ({s.get('executable_rate', 0):.1f}%)") print(f" Average Retries: {s.get('avg_retries', 0):.2f}") print(f" Average Latency: {s.get('avg_latency', 0):.2f}s") print(f"\nšŸ“ RESULTS BY CATEGORY:") for category, stats in s.get("by_category", {}).items(): success_pct = (stats["success"] / stats["total"] * 100) if stats["total"] > 0 else 0 print(f" {category}: {stats['success']}/{stats['total']} ({success_pct:.0f}%)") print(f"\nāŒ ERROR TYPES:") if s.get("failure_types"): for error_type, count in s["failure_types"].items(): print(f" {error_type}: {count}") else: print(" None (all prompts succeeded!)") print(f"\nšŸ’° COST vs QUALITY ANALYSIS:") cost_analysis = self._calculate_cost_analysis() for key, value in cost_analysis.items(): if key != "note": print(f" {key}: {value}") print(f"\nāœ… RECOMMENDATION: {cost_analysis.get('recommendation', 'Unknown')}") print("="*70 + "\n") def run_evaluation_suite(): """Run the complete evaluation suite.""" evaluator = EvaluationFramework(use_llm=False) # Use rule-based for faster testing report = evaluator.run_evaluation(dataset_size="full") evaluator.print_report() return report if __name__ == "__main__": run_evaluation_suite()