File size: 2,505 Bytes
d0fdbcd
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
#!/usr/bin/env python3
"""
Run evaluation suite and generate detailed reports.
"""

import sys
import json
from pathlib import Path
from datetime import datetime

# Add paths
sys.path.insert(0, str(Path(__file__).parent / "src"))
sys.path.insert(0, str(Path(__file__).parent / "evaluation"))

from evaluator import EvaluationFramework


def main():
    """Run evaluation suite."""
    
    print("\n" + "="*80)
    print("πŸ“Š EVALUATION FRAMEWORK - CODE GENERATION SYSTEM")
    print("="*80)
    print("\nRunning comprehensive evaluation on 20 test prompts...")
    print("(10 real products + 10 edge cases)\n")
    
    # Run evaluation
    evaluator = EvaluationFramework(use_llm=False)
    report = evaluator.run_evaluation(dataset_size="full")
    
    # Print formatted report
    evaluator.print_report()
    
    # Save detailed report to file
    timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
    report_file = Path(__file__).parent / f"evaluation_report_{timestamp}.json"
    
    with open(report_file, 'w') as f:
        json.dump(report, f, indent=2, default=str)
    
    print(f"πŸ“ Detailed report saved to: {report_file}\n")
    
    # Print key takeaways
    summary = report["summary"]
    print("\n" + "="*80)
    print("πŸ“ˆ KEY PERFORMANCE INDICATORS")
    print("="*80)
    
    print(f"\nβœ“ Success Rate: {summary.get('success_rate', 0):.1f}%")
    print(f"βœ“ Executable Rate: {summary.get('executable_rate', 0):.1f}%")
    print(f"βœ“ Average Generation Time: {summary.get('avg_latency', 0):.2f}s")
    print(f"βœ“ Average Retries: {summary.get('avg_retries', 0):.1f}")
    
    # Performance by category
    print(f"\nπŸ“ Performance by Category:")
    for category, stats in summary.get("by_category", {}).items():
        success_pct = (stats["success"] / stats["total"] * 100) if stats["total"] > 0 else 0
        print(f"   {category:15} {stats['success']:2}/{stats['total']} ({success_pct:5.1f}%)")
    
    # Cost analysis
    cost = report["cost_analysis"]
    print(f"\nπŸ’° Cost vs Quality Analysis:")
    print(f"   Config Size (avg): {cost.get('avg_config_size_bytes', 0):.0f} bytes")
    print(f"   Latency (avg): {cost.get('avg_generation_latency_seconds', 0):.2f}s")
    print(f"   Quality Score: {cost.get('quality_score', 0):.1f}/100")
    print(f"   Efficiency Score: {cost.get('efficiency_score', 0):.1f}/100")
    print(f"   Recommendation: {cost.get('recommendation', 'N/A')}")
    
    print("\n" + "="*80)


if __name__ == "__main__":
    main()