Download src/pipeline.py from 2008robocode-crypto/code-generation-system: direct link, hf CLI and curl.
- Browser
- Download file 19.1 kB
-
https://huggingface.co/spaces/2008robocode-crypto/code-generation-system/resolve/main/src/pipeline.py
- Command line
-
hf download hf://spaces/2008robocode-crypto/code-generation-system/src/pipeline.py
-
curl -L -o pipeline.py https://huggingface.co/spaces/2008robocode-crypto/code-generation-system/resolve/main/src/pipeline.py
19.1 kB
| """ | |
| Multi-stage generation pipeline orchestrator. | |
| Implements the 4-stage compiler-like system for code generation. | |
| """ | |
| import json | |
| import os | |
| from typing import Any, Dict, List, Optional, Tuple | |
| from datetime import datetime | |
| import re | |
| # Mock LLM calls for now - will be replaced with actual API calls | |
| try: | |
| import anthropic | |
| HAS_ANTHROPIC = True | |
| except ImportError: | |
| HAS_ANTHROPIC = False | |
| from validator import Validator | |
| from repair_engine import RepairEngine | |
| class IntentExtractor: | |
| """Stage 1: Extract structured intent from natural language.""" | |
| def __init__(self, use_llm: bool = True): | |
| self.use_llm = use_llm and HAS_ANTHROPIC | |
| def extract(self, user_prompt: str) -> Dict[str, Any]: | |
| """Extract structured intent from user prompt.""" | |
| if self.use_llm: | |
| return self._extract_with_llm(user_prompt) | |
| else: | |
| return self._extract_pattern_based(user_prompt) | |
| def _extract_pattern_based(self, prompt: str) -> Dict[str, Any]: | |
| """Pattern-based intent extraction (fallback).""" | |
| intent = { | |
| "app_name": self._extract_app_name(prompt), | |
| "app_description": prompt[:200], | |
| "key_features": self._extract_features(prompt), | |
| "user_roles": self._extract_roles(prompt), | |
| "core_entities": self._extract_entities(prompt), | |
| "business_requirements": self._extract_requirements(prompt), | |
| "constraints": self._extract_constraints(prompt), | |
| } | |
| return intent | |
| def _extract_app_name(self, prompt: str) -> str: | |
| """Extract app name from prompt.""" | |
| # Look for "Build a X" or "Create a X" | |
| match = re.search(r'(?:Build|Create|Make|Generate)\s+(?:a\s+)?([A-Z][a-zA-Z\s]+?)(?:\s+with|\s+that|\.|$)', prompt) | |
| if match: | |
| return match.group(1).strip().replace(" ", "") | |
| return "GeneratedApp" | |
| def _extract_features(self, prompt: str) -> List[str]: | |
| """Extract key features.""" | |
| features = [] | |
| # Common feature keywords | |
| feature_keywords = [ | |
| "login", "authentication", "contacts", "dashboard", "analytics", | |
| "admin", "payments", "role-based", "access", "premium", "plan", | |
| "reports", "export", "import", "notifications", "search" | |
| ] | |
| for keyword in feature_keywords: | |
| if keyword.lower() in prompt.lower(): | |
| features.append(keyword) | |
| return features or ["basic_crud"] | |
| def _extract_roles(self, prompt: str) -> List[str]: | |
| """Extract user roles.""" | |
| roles = [] | |
| role_keywords = {"admin": "admin", "user": "user", "guest": "guest", "customer": "user"} | |
| for keyword, role in role_keywords.items(): | |
| if keyword.lower() in prompt.lower(): | |
| roles.append(role) | |
| return roles or ["user"] | |
| def _extract_entities(self, prompt: str) -> List[str]: | |
| """Extract core data entities.""" | |
| entities = [] | |
| entity_keywords = { | |
| "contact": "Contact", | |
| "user": "User", | |
| "product": "Product", | |
| "order": "Order", | |
| "payment": "Payment", | |
| "report": "Report", | |
| "dashboard": "Dashboard", | |
| } | |
| for keyword, entity in entity_keywords.items(): | |
| if keyword.lower() in prompt.lower(): | |
| entities.append(entity) | |
| return entities or ["Item"] | |
| def _extract_requirements(self, prompt: str) -> List[str]: | |
| """Extract business requirements.""" | |
| return [ | |
| "User authentication and authorization", | |
| "Role-based access control", | |
| "Data persistence", | |
| "API endpoints for CRUD operations", | |
| ] | |
| def _extract_constraints(self, prompt: str) -> List[str]: | |
| """Extract constraints.""" | |
| constraints = [] | |
| if "premium" in prompt.lower(): | |
| constraints.append("Payment processing required") | |
| if "real-time" in prompt.lower(): | |
| constraints.append("Real-time synchronization needed") | |
| return constraints | |
| def _extract_with_llm(self, prompt: str) -> Dict[str, Any]: | |
| """Extract intent using Anthropic API.""" | |
| try: | |
| client = anthropic.Anthropic(api_key=os.environ.get("ANTHROPIC_API_KEY")) | |
| extraction_prompt = f"""Extract structured intent from this user prompt: | |
| "{prompt}" | |
| Return a JSON with these fields: | |
| - app_name: string (extract or generate a name) | |
| - app_description: string (2-3 sentences) | |
| - key_features: list[string] (extracted features) | |
| - user_roles: list[string] (roles mentioned) | |
| - core_entities: list[string] (data models) | |
| - business_requirements: list[string] (business rules) | |
| - constraints: list[string] (any constraints mentioned) | |
| Return ONLY valid JSON, no markdown formatting.""" | |
| message = client.messages.create( | |
| model="claude-3-5-sonnet-20241022", | |
| max_tokens=1024, | |
| messages=[{"role": "user", "content": extraction_prompt}] | |
| ) | |
| response_text = message.content[0].text | |
| return json.loads(response_text) | |
| except Exception as e: | |
| print(f"LLM extraction failed: {e}, falling back to pattern-based") | |
| return self._extract_pattern_based(prompt) | |
| class SystemDesignLayer: | |
| """Stage 2: Convert intent to system design.""" | |
| def __init__(self, use_llm: bool = True): | |
| self.use_llm = use_llm and HAS_ANTHROPIC | |
| def design(self, intent: Dict[str, Any]) -> Dict[str, Any]: | |
| """Generate system design from intent.""" | |
| if self.use_llm: | |
| return self._design_with_llm(intent) | |
| else: | |
| return self._design_rule_based(intent) | |
| def _design_rule_based(self, intent: Dict[str, Any]) -> Dict[str, Any]: | |
| """Rule-based system design.""" | |
| design = { | |
| "entities": self._generate_entities(intent), | |
| "user_flows": self._generate_flows(intent), | |
| "roles_and_permissions": self._generate_rbac(intent), | |
| "data_models": intent["core_entities"], | |
| "api_patterns": ["REST"], | |
| "ui_structure": self._generate_ui_structure(intent), | |
| } | |
| return design | |
| def _generate_entities(self, intent: Dict[str, Any]) -> Dict[str, List[str]]: | |
| """Generate entity definitions.""" | |
| entities = {} | |
| for entity in intent["core_entities"]: | |
| if entity.lower() == "user": | |
| entities[entity] = ["id", "name", "email", "role", "created_at"] | |
| elif entity.lower() == "contact": | |
| entities[entity] = ["id", "name", "email", "phone", "owner_id"] | |
| elif entity.lower() == "product": | |
| entities[entity] = ["id", "name", "price", "description"] | |
| elif entity.lower() == "order": | |
| entities[entity] = ["id", "user_id", "total", "status", "created_at"] | |
| else: | |
| entities[entity] = ["id", "name", "created_at"] | |
| return entities | |
| def _generate_flows(self, intent: Dict[str, Any]) -> List[Dict[str, Any]]: | |
| """Generate user flows.""" | |
| flows = [ | |
| {"name": "Authentication", "steps": ["Login", "Verify", "Redirect to Dashboard"]}, | |
| {"name": "CRUD Operations", "steps": ["View", "Create", "Update", "Delete"]}, | |
| ] | |
| if "admin" in intent["user_roles"]: | |
| flows.append({"name": "Admin Panel", "steps": ["View Analytics", "Manage Users", "View Reports"]}) | |
| return flows | |
| def _generate_rbac(self, intent: Dict[str, Any]) -> Dict[str, List[str]]: | |
| """Generate role-based access control.""" | |
| rbac = {} | |
| for role in intent["user_roles"]: | |
| if role == "admin": | |
| rbac[role] = ["read_all", "write_all", "delete_all", "manage_users"] | |
| elif role == "user": | |
| rbac[role] = ["read_own", "write_own", "delete_own"] | |
| else: | |
| rbac[role] = ["read_public"] | |
| return rbac | |
| def _generate_ui_structure(self, intent: Dict[str, Any]) -> List[str]: | |
| """Generate UI page structure.""" | |
| pages = ["/login", "/dashboard", "/profile"] | |
| if "contacts" in str(intent["key_features"]).lower(): | |
| pages.append("/contacts") | |
| if "admin" in intent["user_roles"]: | |
| pages.append("/admin") | |
| if "analytics" in str(intent["key_features"]).lower(): | |
| pages.append("/analytics") | |
| return pages | |
| def _design_with_llm(self, intent: Dict[str, Any]) -> Dict[str, Any]: | |
| """Generate system design using LLM.""" | |
| try: | |
| client = anthropic.Anthropic(api_key=os.environ.get("ANTHROPIC_API_KEY")) | |
| design_prompt = f"""Design a system architecture based on this intent: | |
| {json.dumps(intent, indent=2)} | |
| Return a JSON with these fields: | |
| - entities: dict mapping entity names to attribute lists | |
| - user_flows: list of flow objects with name and steps | |
| - roles_and_permissions: dict mapping roles to permissions | |
| - data_models: list of entity names | |
| - api_patterns: list (e.g., ["REST", "GraphQL"]) | |
| - ui_structure: list of page paths | |
| Return ONLY valid JSON.""" | |
| message = client.messages.create( | |
| model="claude-3-5-sonnet-20241022", | |
| max_tokens=2048, | |
| messages=[{"role": "user", "content": design_prompt}] | |
| ) | |
| response_text = message.content[0].text | |
| return json.loads(response_text) | |
| except Exception as e: | |
| print(f"LLM design failed: {e}, using rule-based") | |
| return self._design_rule_based(intent) | |
| class SchemaGenerator: | |
| """Stage 3: Generate complete schemas (DB, API, UI, Auth).""" | |
| def __init__(self, use_llm: bool = True): | |
| self.use_llm = use_llm and HAS_ANTHROPIC | |
| def generate(self, design: Dict[str, Any], intent: Dict[str, Any]) -> Dict[str, Any]: | |
| """Generate complete schema from design.""" | |
| if self.use_llm: | |
| return self._generate_with_llm(design, intent) | |
| else: | |
| return self._generate_rule_based(design, intent) | |
| def _generate_rule_based(self, design: Dict[str, Any], intent: Dict[str, Any]) -> Dict[str, Any]: | |
| """Rule-based schema generation.""" | |
| schema = { | |
| "app_name": intent["app_name"], | |
| "app_description": intent["app_description"], | |
| "database_schema": self._generate_db_schema(design), | |
| "api_schema": self._generate_api_schema(design), | |
| "ui_schema": self._generate_ui_schema(design), | |
| "auth_config": self._generate_auth_config(design), | |
| "roles": self._generate_roles(design), | |
| "business_logic": self._generate_business_logic(intent), | |
| } | |
| return schema | |
| def _generate_db_schema(self, design: Dict[str, Any]) -> List[Dict[str, Any]]: | |
| """Generate database schema.""" | |
| tables = [] | |
| for entity, attributes in design.get("entities", {}).items(): | |
| table = { | |
| "name": entity.lower() + "s", | |
| "fields": [ | |
| {"name": "id", "type": "string", "required": True}, | |
| ] + [ | |
| {"name": attr, "type": "string", "required": True} | |
| for attr in attributes if attr != "id" | |
| ], | |
| "primary_key": "id", | |
| "indexes": ["id"] | |
| } | |
| tables.append(table) | |
| return tables | |
| def _generate_api_schema(self, design: Dict[str, Any]) -> List[Dict[str, Any]]: | |
| """Generate API schema.""" | |
| endpoints = [] | |
| for entity in design.get("data_models", []): | |
| base_path = f"/api/{entity.lower()}s" | |
| endpoints.extend([ | |
| {"path": base_path, "method": "GET", "description": f"List {entity}s"}, | |
| {"path": f"{base_path}/{{id}}", "method": "GET", "description": f"Get {entity}"}, | |
| {"path": base_path, "method": "POST", "description": f"Create {entity}"}, | |
| {"path": f"{base_path}/{{id}}", "method": "PUT", "description": f"Update {entity}"}, | |
| {"path": f"{base_path}/{{id}}", "method": "DELETE", "description": f"Delete {entity}"}, | |
| ]) | |
| return endpoints | |
| def _generate_ui_schema(self, design: Dict[str, Any]) -> List[Dict[str, Any]]: | |
| """Generate UI schema.""" | |
| pages = [] | |
| for path in design.get("ui_structure", []): | |
| page = { | |
| "path": path, | |
| "title": path.replace("/", " ").title(), | |
| "components": [ | |
| {"name": "header", "type": "header"}, | |
| {"name": "content", "type": "container"}, | |
| {"name": "footer", "type": "footer"}, | |
| ] | |
| } | |
| pages.append(page) | |
| return pages | |
| def _generate_auth_config(self, design: Dict[str, Any]) -> Dict[str, Any]: | |
| """Generate authentication config.""" | |
| return { | |
| "type": "jwt", | |
| "secret_key": "generated-secret", | |
| "expiry": 3600, | |
| "refresh_token_expiry": 86400, | |
| } | |
| def _generate_roles(self, design: Dict[str, Any]) -> List[Dict[str, Any]]: | |
| """Generate roles from RBAC.""" | |
| roles = [] | |
| for role_name, permissions in design.get("roles_and_permissions", {}).items(): | |
| roles.append({ | |
| "name": role_name, | |
| "permissions": permissions, | |
| "description": f"Role: {role_name}" | |
| }) | |
| return roles | |
| def _generate_business_logic(self, intent: Dict[str, Any]) -> Dict[str, Any]: | |
| """Generate business logic rules.""" | |
| logic = { | |
| "validation_rules": [ | |
| "Email must be valid format", | |
| "Password must be at least 8 characters", | |
| ], | |
| "access_control": "Role-based access control enabled", | |
| "premium_features": "premium" in str(intent).lower(), | |
| } | |
| return logic | |
| def _generate_with_llm(self, design: Dict[str, Any], intent: Dict[str, Any]) -> Dict[str, Any]: | |
| """Generate schemas using LLM.""" | |
| try: | |
| client = anthropic.Anthropic(api_key=os.environ.get("ANTHROPIC_API_KEY")) | |
| schema_prompt = f"""Generate complete schemas from this design and intent: | |
| Design: {json.dumps(design, indent=2)} | |
| Intent: {json.dumps(intent, indent=2)} | |
| Return a JSON with these fields: | |
| - app_name: string | |
| - app_description: string | |
| - database_schema: list of tables (each with name, fields, primary_key) | |
| - api_schema: list of endpoints (path, method, description) | |
| - ui_schema: list of pages (path, title, components) | |
| - auth_config: object with type, expiry, etc. | |
| - roles: list of role objects (name, permissions, description) | |
| - business_logic: object with business rules | |
| All table fields must be objects with: name, type, required | |
| Valid types: string, number, boolean, date, email, enum | |
| All API endpoints must have valid HTTP methods: GET, POST, PUT, DELETE | |
| Return ONLY valid JSON.""" | |
| message = client.messages.create( | |
| model="claude-3-5-sonnet-20241022", | |
| max_tokens=4096, | |
| messages=[{"role": "user", "content": schema_prompt}] | |
| ) | |
| response_text = message.content[0].text | |
| return json.loads(response_text) | |
| except Exception as e: | |
| print(f"LLM schema generation failed: {e}, using rule-based") | |
| return self._generate_rule_based(design, intent) | |
| class RefinementLayer: | |
| """Stage 4: Refine and validate schemas across all layers.""" | |
| def __init__(self): | |
| self.validator = Validator() | |
| self.repair_engine = RepairEngine() | |
| def refine(self, schema: Dict[str, Any], max_iterations: int = 3) -> Tuple[Dict[str, Any], Dict[str, Any]]: | |
| """Validate and repair schema iteratively.""" | |
| metadata = { | |
| "iterations": 0, | |
| "validation_results": [], | |
| "repairs": [], | |
| "final_status": "unknown", | |
| } | |
| for i in range(max_iterations): | |
| metadata["iterations"] = i + 1 | |
| # Validate | |
| result = self.validator.validate_complete(schema) | |
| metadata["validation_results"].append(result.to_dict()) | |
| if result.is_valid: | |
| metadata["final_status"] = "valid" | |
| return schema, metadata | |
| # Repair | |
| schema, repairs = self.repair_engine.repair_config(schema) | |
| metadata["repairs"].extend(repairs) | |
| metadata["final_status"] = "repaired_with_warnings" if metadata["validation_results"][-1]["errors"] else "valid" | |
| return schema, metadata | |
| class Pipeline: | |
| """Main orchestrator for the 4-stage pipeline.""" | |
| def __init__(self, use_llm: bool = True): | |
| self.intent_extractor = IntentExtractor(use_llm=use_llm) | |
| self.system_design = SystemDesignLayer(use_llm=use_llm) | |
| self.schema_generator = SchemaGenerator(use_llm=use_llm) | |
| self.refinement = RefinementLayer() | |
| self.use_llm = use_llm | |
| def generate(self, user_prompt: str) -> Tuple[Dict[str, Any], Dict[str, Any]]: | |
| """Run complete pipeline: prompt → config.""" | |
| execution_log = { | |
| "timestamp": datetime.now().isoformat(), | |
| "user_prompt": user_prompt[:500], | |
| "stages": {} | |
| } | |
| try: | |
| # Stage 1: Intent Extraction | |
| intent = self.intent_extractor.extract(user_prompt) | |
| execution_log["stages"]["intent_extraction"] = {"status": "completed"} | |
| # Stage 2: System Design | |
| design = self.system_design.design(intent) | |
| execution_log["stages"]["system_design"] = {"status": "completed"} | |
| # Stage 3: Schema Generation | |
| schema = self.schema_generator.generate(design, intent) | |
| execution_log["stages"]["schema_generation"] = {"status": "completed"} | |
| # Stage 4: Refinement | |
| refined_schema, refinement_metadata = self.refinement.refine(schema) | |
| execution_log["stages"]["refinement"] = refinement_metadata | |
| execution_log["final_status"] = "success" | |
| return refined_schema, execution_log | |
| except Exception as e: | |
| execution_log["final_status"] = "error" | |
| execution_log["error"] = str(e) | |
| return {}, execution_log | |