Download evaluation/evaluator.py from 2008robocode-crypto/code-generation-system: direct link, hf CLI and curl.
- Browser
- Download file 10.3 kB
-
https://huggingface.co/spaces/2008robocode-crypto/code-generation-system/resolve/main/evaluation/evaluator.py
- Command line
-
hf download hf://spaces/2008robocode-crypto/code-generation-system/evaluation/evaluator.py
-
curl -L -o evaluator.py https://huggingface.co/spaces/2008robocode-crypto/code-generation-system/resolve/main/evaluation/evaluator.py
10.3 kB
| """ | |
| Evaluation framework to measure system performance on test dataset. | |
| Tracks success rate, retries, failure types, and latency. | |
| """ | |
| import json | |
| import time | |
| from typing import Any, Dict, List, Optional | |
| from datetime import datetime | |
| import sys | |
| import os | |
| # Add src to path | |
| sys.path.insert(0, os.path.join(os.path.dirname(__file__), '..', 'src')) | |
| from pipeline import Pipeline | |
| from runtime_simulator import validate_config_executable | |
| from test_dataset import get_test_dataset, get_real_prompts, get_edge_cases | |
| class EvaluationFramework: | |
| """Comprehensive evaluation of the system.""" | |
| def __init__(self, use_llm: bool = True): | |
| self.pipeline = Pipeline(use_llm=use_llm) | |
| self.results = [] | |
| self.summary = { | |
| "total_prompts": 0, | |
| "successful": 0, | |
| "failed": 0, | |
| "executable": 0, | |
| "total_retries": 0, | |
| "total_latency": 0.0, | |
| "by_category": {}, | |
| "failure_types": {}, | |
| "timestamp": datetime.now().isoformat(), | |
| } | |
| def evaluate_prompt(self, prompt: Dict[str, Any], max_retries: int = 1) -> Dict[str, Any]: | |
| """Evaluate a single prompt.""" | |
| prompt_id = prompt.get("id", "unknown") | |
| prompt_text = prompt.get("prompt", "") | |
| category = prompt.get("category", prompt.get("type", "unknown")) | |
| result = { | |
| "prompt_id": prompt_id, | |
| "category": category, | |
| "prompt_summary": prompt_text[:100], | |
| "success": False, | |
| "executable": False, | |
| "retries": 0, | |
| "latency": 0.0, | |
| "errors": [], | |
| "warnings": [], | |
| "config_size": 0, | |
| } | |
| start_time = time.time() | |
| # Try generation with retries | |
| for attempt in range(max_retries): | |
| result["retries"] = attempt + 1 | |
| try: | |
| config, exec_log = self.pipeline.generate(prompt_text) | |
| if not config: | |
| result["errors"].append("Empty config generated") | |
| continue | |
| # Check if executable | |
| is_executable, exec_report = validate_config_executable(config) | |
| result["success"] = True | |
| result["executable"] = is_executable | |
| result["config_size"] = len(json.dumps(config)) | |
| if not is_executable: | |
| result["warnings"].extend(exec_report.get("warnings", [])) | |
| result["errors"].extend(exec_report.get("errors", [])) | |
| # Store execution log | |
| result["execution_log"] = exec_log | |
| result["execution_report"] = exec_report | |
| break | |
| except Exception as e: | |
| error_msg = str(e) | |
| result["errors"].append(error_msg) | |
| # Categorize error | |
| error_type = self._categorize_error(error_msg) | |
| if error_type not in self.summary["failure_types"]: | |
| self.summary["failure_types"][error_type] = 0 | |
| self.summary["failure_types"][error_type] += 1 | |
| result["latency"] = time.time() - start_time | |
| return result | |
| def _categorize_error(self, error: str) -> str: | |
| """Categorize error type.""" | |
| error_lower = error.lower() | |
| if "json" in error_lower: | |
| return "json_error" | |
| elif "validation" in error_lower: | |
| return "validation_error" | |
| elif "field" in error_lower: | |
| return "field_error" | |
| elif "api" in error_lower: | |
| return "api_error" | |
| elif "database" in error_lower or "table" in error_lower: | |
| return "database_error" | |
| else: | |
| return "unknown_error" | |
| def run_evaluation(self, dataset_size: str = "full") -> Dict[str, Any]: | |
| """Run full evaluation on test dataset.""" | |
| if dataset_size == "full": | |
| prompts = get_real_prompts() + get_edge_cases() | |
| elif dataset_size == "real": | |
| prompts = get_real_prompts() | |
| elif dataset_size == "edge": | |
| prompts = get_edge_cases() | |
| else: | |
| prompts = get_real_prompts()[:int(dataset_size)] | |
| self.summary["total_prompts"] = len(prompts) | |
| print(f"\n๐ Running evaluation on {len(prompts)} prompts...") | |
| for i, prompt in enumerate(prompts): | |
| print(f" [{i+1}/{len(prompts)}] {prompt.get('name', prompt.get('id'))}", end=" ") | |
| result = self.evaluate_prompt(prompt) | |
| self.results.append(result) | |
| # Update summary | |
| if result["success"]: | |
| self.summary["successful"] += 1 | |
| print("โ") | |
| else: | |
| self.summary["failed"] += 1 | |
| print("โ") | |
| if result["executable"]: | |
| self.summary["executable"] += 1 | |
| self.summary["total_retries"] += result["retries"] | |
| self.summary["total_latency"] += result["latency"] | |
| # Track by category | |
| category = result["category"] | |
| if category not in self.summary["by_category"]: | |
| self.summary["by_category"][category] = {"success": 0, "total": 0} | |
| self.summary["by_category"][category]["total"] += 1 | |
| if result["success"]: | |
| self.summary["by_category"][category]["success"] += 1 | |
| # Calculate metrics | |
| self.summary["success_rate"] = (self.summary["successful"] / self.summary["total_prompts"]) * 100 if self.summary["total_prompts"] > 0 else 0 | |
| self.summary["executable_rate"] = (self.summary["executable"] / self.summary["total_prompts"]) * 100 if self.summary["total_prompts"] > 0 else 0 | |
| self.summary["avg_retries"] = self.summary["total_retries"] / self.summary["total_prompts"] if self.summary["total_prompts"] > 0 else 0 | |
| self.summary["avg_latency"] = self.summary["total_latency"] / self.summary["total_prompts"] if self.summary["total_prompts"] > 0 else 0 | |
| return self.get_report() | |
| def get_report(self) -> Dict[str, Any]: | |
| """Generate evaluation report.""" | |
| return { | |
| "summary": self.summary, | |
| "detailed_results": self.results, | |
| "cost_analysis": self._calculate_cost_analysis(), | |
| } | |
| def _calculate_cost_analysis(self) -> Dict[str, Any]: | |
| """Analyze cost vs quality tradeoff.""" | |
| if not self.results: | |
| return {} | |
| successful_configs = [r for r in self.results if r["success"]] | |
| if not successful_configs: | |
| return {"note": "No successful generations to analyze"} | |
| avg_config_size = sum(r["config_size"] for r in successful_configs) / len(successful_configs) | |
| avg_latency = sum(r["latency"] for r in successful_configs) / len(successful_configs) | |
| return { | |
| "avg_config_size_bytes": avg_config_size, | |
| "avg_generation_latency_seconds": round(avg_latency, 2), | |
| "estimated_api_calls_per_prompt": 4, # 4 stages | |
| "estimated_tokens_per_generation": int(avg_config_size / 4), # Rough estimate | |
| "quality_score": (self.summary["success_rate"] * 0.6) + (self.summary["executable_rate"] * 0.4), | |
| "efficiency_score": 100 - (avg_latency * 10), # Arbitrary scale | |
| "recommendation": self._get_recommendation(), | |
| } | |
| def _get_recommendation(self) -> str: | |
| """Get recommendation based on metrics.""" | |
| success_rate = self.summary.get("success_rate", 0) | |
| executable_rate = self.summary.get("executable_rate", 0) | |
| if success_rate >= 80 and executable_rate >= 75: | |
| return "Production-ready with monitoring" | |
| elif success_rate >= 60 and executable_rate >= 50: | |
| return "Ready for limited production use" | |
| elif success_rate >= 40: | |
| return "Needs refinement before production" | |
| else: | |
| return "Requires significant improvements" | |
| def print_report(self): | |
| """Print formatted report.""" | |
| print("\n" + "="*70) | |
| print("๐ EVALUATION REPORT") | |
| print("="*70) | |
| s = self.summary | |
| print(f"\n๐ SUMMARY METRICS:") | |
| print(f" Total Prompts Evaluated: {s['total_prompts']}") | |
| print(f" Successful Generations: {s['successful']}/{s['total_prompts']} ({s.get('success_rate', 0):.1f}%)") | |
| print(f" Executable Configs: {s['executable']}/{s['total_prompts']} ({s.get('executable_rate', 0):.1f}%)") | |
| print(f" Average Retries: {s.get('avg_retries', 0):.2f}") | |
| print(f" Average Latency: {s.get('avg_latency', 0):.2f}s") | |
| print(f"\n๐ RESULTS BY CATEGORY:") | |
| for category, stats in s.get("by_category", {}).items(): | |
| success_pct = (stats["success"] / stats["total"] * 100) if stats["total"] > 0 else 0 | |
| print(f" {category}: {stats['success']}/{stats['total']} ({success_pct:.0f}%)") | |
| print(f"\nโ ERROR TYPES:") | |
| if s.get("failure_types"): | |
| for error_type, count in s["failure_types"].items(): | |
| print(f" {error_type}: {count}") | |
| else: | |
| print(" None (all prompts succeeded!)") | |
| print(f"\n๐ฐ COST vs QUALITY ANALYSIS:") | |
| cost_analysis = self._calculate_cost_analysis() | |
| for key, value in cost_analysis.items(): | |
| if key != "note": | |
| print(f" {key}: {value}") | |
| print(f"\nโ RECOMMENDATION: {cost_analysis.get('recommendation', 'Unknown')}") | |
| print("="*70 + "\n") | |
| def run_evaluation_suite(): | |
| """Run the complete evaluation suite.""" | |
| evaluator = EvaluationFramework(use_llm=False) # Use rule-based for faster testing | |
| report = evaluator.run_evaluation(dataset_size="full") | |
| evaluator.print_report() | |
| return report | |
| if __name__ == "__main__": | |
| run_evaluation_suite() | |