Download run_evaluation.py from 2008robocode-crypto/code-generation-system: direct link, hf CLI and curl.
- Browser
- Download file 2.51 kB
-
https://huggingface.co/spaces/2008robocode-crypto/code-generation-system/resolve/main/run_evaluation.py
- Command line
-
hf download hf://spaces/2008robocode-crypto/code-generation-system/run_evaluation.py
-
curl -L -o run_evaluation.py https://huggingface.co/spaces/2008robocode-crypto/code-generation-system/resolve/main/run_evaluation.py
2.51 kB
| #!/usr/bin/env python3 | |
| """ | |
| Run evaluation suite and generate detailed reports. | |
| """ | |
| import sys | |
| import json | |
| from pathlib import Path | |
| from datetime import datetime | |
| # Add paths | |
| sys.path.insert(0, str(Path(__file__).parent / "src")) | |
| sys.path.insert(0, str(Path(__file__).parent / "evaluation")) | |
| from evaluator import EvaluationFramework | |
| def main(): | |
| """Run evaluation suite.""" | |
| print("\n" + "="*80) | |
| print("π EVALUATION FRAMEWORK - CODE GENERATION SYSTEM") | |
| print("="*80) | |
| print("\nRunning comprehensive evaluation on 20 test prompts...") | |
| print("(10 real products + 10 edge cases)\n") | |
| # Run evaluation | |
| evaluator = EvaluationFramework(use_llm=False) | |
| report = evaluator.run_evaluation(dataset_size="full") | |
| # Print formatted report | |
| evaluator.print_report() | |
| # Save detailed report to file | |
| timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") | |
| report_file = Path(__file__).parent / f"evaluation_report_{timestamp}.json" | |
| with open(report_file, 'w') as f: | |
| json.dump(report, f, indent=2, default=str) | |
| print(f"π Detailed report saved to: {report_file}\n") | |
| # Print key takeaways | |
| summary = report["summary"] | |
| print("\n" + "="*80) | |
| print("π KEY PERFORMANCE INDICATORS") | |
| print("="*80) | |
| print(f"\nβ Success Rate: {summary.get('success_rate', 0):.1f}%") | |
| print(f"β Executable Rate: {summary.get('executable_rate', 0):.1f}%") | |
| print(f"β Average Generation Time: {summary.get('avg_latency', 0):.2f}s") | |
| print(f"β Average Retries: {summary.get('avg_retries', 0):.1f}") | |
| # Performance by category | |
| print(f"\nπ Performance by Category:") | |
| for category, stats in summary.get("by_category", {}).items(): | |
| success_pct = (stats["success"] / stats["total"] * 100) if stats["total"] > 0 else 0 | |
| print(f" {category:15} {stats['success']:2}/{stats['total']} ({success_pct:5.1f}%)") | |
| # Cost analysis | |
| cost = report["cost_analysis"] | |
| print(f"\nπ° Cost vs Quality Analysis:") | |
| print(f" Config Size (avg): {cost.get('avg_config_size_bytes', 0):.0f} bytes") | |
| print(f" Latency (avg): {cost.get('avg_generation_latency_seconds', 0):.2f}s") | |
| print(f" Quality Score: {cost.get('quality_score', 0):.1f}/100") | |
| print(f" Efficiency Score: {cost.get('efficiency_score', 0):.1f}/100") | |
| print(f" Recommendation: {cost.get('recommendation', 'N/A')}") | |
| print("\n" + "="*80) | |
| if __name__ == "__main__": | |
| main() | |