Spaces:
Sleeping
Sleeping
| # main.py β AutoReview AI | |
| # LangChain-based autonomous GitHub PR reviewer | |
| import os | |
| import re | |
| from dotenv import load_dotenv | |
| from langchain_groq import ChatGroq | |
| from langchain.agents import AgentExecutor, create_tool_calling_agent | |
| from langchain_core.prompts import ChatPromptTemplate | |
| from langchain_core.messages import HumanMessage, SystemMessage | |
| from tools import fetch_pr_diff, analyze_diff, post_inline_comment, post_overall_review, fetch_pr_data, format_pr_data | |
| load_dotenv() | |
| # ββ LLM ββββββββββββββββββββββββββββββββββ | |
| llm = ChatGroq(model="openai/gpt-oss-120b", temperature=0) | |
| # ββ TOOLS βββββββββββββββββββββββββββββββββ | |
| tools = [fetch_pr_diff, analyze_diff, post_inline_comment, post_overall_review] | |
| # ββ PROMPT ββββββββββββββββββββββββββββββββ | |
| prompt = ChatPromptTemplate.from_messages([ | |
| ("system", """You are AutoReview AI β an expert autonomous GitHub PR reviewer. | |
| Your job when given a PR URL: | |
| 1. Use fetch_pr_diff to get the PR details and code changes | |
| 2. Use analyze_diff for each changed file to find bugs, security issues, style problems | |
| 3. Use post_inline_comment to post specific comments on problematic lines | |
| 4. Use post_overall_review to post a final APPROVE or REQUEST_CHANGES verdict | |
| Be thorough but concise. Focus on real issues that matter. | |
| Always complete all 4 steps β don't stop after fetching. | |
| If GitHub token is not available, still analyze and return the review as text. | |
| """), | |
| ("human", "{input}"), | |
| ("placeholder", "{agent_scratchpad}") | |
| ]) | |
| # ββ AGENT βββββββββββββββββββββββββββββββββ | |
| agent = create_tool_calling_agent(llm, tools, prompt) | |
| agent_executor = AgentExecutor( | |
| agent=agent, | |
| tools=tools, | |
| verbose=True, | |
| max_iterations=10, | |
| handle_parsing_errors=True | |
| ) | |
| # βββββββββββββββββββββββββββββββββββββββββ | |
| # MAIN REVIEW FUNCTION | |
| # βββββββββββββββββββββββββββββββββββββββββ | |
| def review_pr( | |
| pr_url: str, | |
| github_token: str | None = None | |
| ) -> dict: | |
| """ | |
| Main function to review a GitHub PR. | |
| Returns structured review result. | |
| """ | |
| print(f"\n{'='*60}") | |
| print(f"π AutoReview AI starting review...") | |
| print(f"π PR: {pr_url}") | |
| print("="*60) | |
| try: | |
| result = agent_executor.invoke({ | |
| "input": f""" | |
| Please review this GitHub Pull Request: {pr_url} | |
| GitHub Token: {github_token if github_token else 'Not Provided'} | |
| Steps to follow: | |
| 1. Fetch the PR diff using fetch_pr_diff | |
| 2. Analyze each changed file using analyze_diff | |
| 3. Post inline comments for specific issues using post_inline_comment | |
| 4. Post overall verdict using post_overall_review | |
| Provide a thorough code review focusing on: | |
| - Bugs and logic errors | |
| - Security vulnerabilities | |
| - Code style and best practices | |
| - Performance issues | |
| """ | |
| }) | |
| output = result.get("output", "") | |
| print(f"\nβ Review complete!") | |
| print(f"\nOutput:\n{output}") | |
| return { | |
| "success": True, | |
| "pr_url": pr_url, | |
| "review": output, | |
| "error": None | |
| } | |
| except Exception as e: | |
| error_msg = str(e) | |
| print(f"\nβ Error: {error_msg}") | |
| return { | |
| "success": False, | |
| "pr_url": pr_url, | |
| "review": None, | |
| "error": error_msg | |
| } | |
| # βββββββββββββββββββββββββββββββββββββββββ | |
| # SIMPLE REVIEW (no GitHub token needed) | |
| # For demo purposes | |
| # βββββββββββββββββββββββββββββββββββββββββ | |
| def review_pr_simple( | |
| pr_url: str, | |
| github_token: str | None = None | |
| ) -> dict: | |
| """ | |
| Reviews PR without posting comments to GitHub. | |
| Analyzes each changed file separately (instead of truncating the | |
| whole diff to one 4000-char blob), and never lets a Groq/GitHub | |
| error crash the caller β always returns a usable dict. | |
| """ | |
| print(f"\nπ Fetching PR data...") | |
| try: | |
| # Step 1: Fetch structured PR data (real file list, not a string) | |
| data = fetch_pr_data( | |
| pr_url=pr_url, | |
| github_token=github_token | |
| ) | |
| if not data["success"]: | |
| return {"success": False, "error": data["error"], "review": None} | |
| print(f"β PR fetched! {len(data['files'])} file(s) changed") | |
| # Step 2: Analyze EACH file separately (cap at 8 files for demo speed) | |
| per_file_analysis = [] | |
| files_to_review = data["files"][:8] | |
| for f in files_to_review: | |
| print(f"π€ Analyzing {f['filename']}...") | |
| file_block = ( | |
| f"--- {f['filename']} ({f['status']}) " | |
| f"+{f['additions']}/-{f['deletions']} ---\n{f['patch']}" | |
| ) | |
| try: | |
| file_analysis = analyze_diff.invoke(file_block) | |
| except Exception as e: | |
| file_analysis = f"ERROR analyzing this file: {str(e)}" | |
| per_file_analysis.append(f"## {f['filename']}\n{file_analysis}") | |
| if not per_file_analysis: | |
| analysis = "No file changes with diffs to analyze." | |
| else: | |
| analysis = "\n\n".join(per_file_analysis) | |
| print("β Analysis complete!") | |
| # Step 3: Generate final verdict from the per-file analyses | |
| try: | |
| verdict_response = llm.invoke([ | |
| SystemMessage("You are a senior engineer. Give a final PR review verdict."), | |
| HumanMessage(f""" | |
| Based on this PR analysis, write a complete review report: | |
| PR TITLE: {data['pr_title']} | |
| PR DESCRIPTION: {data['pr_description'][:500]} | |
| FILES CHANGED: {data['total_files']} | |
| PER-FILE ANALYSIS: | |
| {analysis[:6000]} | |
| Write a professional review with: | |
| 1. Overall verdict (APPROVE / REQUEST CHANGES) | |
| 2. Key issues found | |
| 3. Positive aspects | |
| 4. Specific suggestions | |
| """) | |
| ]) | |
| verdict = verdict_response.content | |
| except Exception as e: | |
| verdict = f"Could not generate final verdict β LLM error: {str(e)}" | |
| return { | |
| "success": True, | |
| "pr_url": pr_url, | |
| "pr_data": format_pr_data(data), | |
| "analysis": analysis, | |
| "verdict": verdict, | |
| "error": None | |
| } | |
| except Exception as e: | |
| # Catch-all so an unexpected error (network, parsing, anything) | |
| # never crashes app.py β it always gets a usable dict back | |
| return {"success": False, "error": f"Unexpected error: {str(e)}", "review": None} | |
| # βββββββββββββββββββββββββββββββββββββββββ | |
| # RUN | |
| # βββββββββββββββββββββββββββββββββββββββββ | |
| if __name__ == "__main__": | |
| # Test with a real public PR | |
| test_pr = "https://github.com/langchain-ai/langchain/pull/1" | |
| print("Mode: Simple review (no GitHub token needed for demo)") | |
| result = review_pr_simple(test_pr, None) | |
| if result["success"]: | |
| print(f"\n{'='*60}") | |
| print("π PR DATA SUMMARY:") | |
| print(result["pr_data"][:500]) | |
| print(f"\nπ ANALYSIS:") | |
| print(result["analysis"]) | |
| print(f"\nβ FINAL VERDICT:") | |
| print(result["verdict"]) | |
| else: | |
| print(f"β Failed: {result['error']}") |