import os import re import json import time import tempfile from pathlib import Path import gradio as gr import pandas as pd import requests from smolagents import CodeAgent, tool, DuckDuckGoSearchTool, VisitWebpageTool try: from smolagents import OpenAIServerModel except ImportError: from smolagents import OpenAIModel as OpenAIServerModel API_URL = "https://agents-course-unit4-scoring.hf.space" @tool def get_task_file_url(task_id: str) -> str: """ Build the file URL for a GAIA task. Args: task_id: The task identifier. Returns: The download URL for the task file. """ return f"{API_URL}/files/{task_id}" @tool def download_file(url: str) -> str: """ Download a remote file to a temporary local path. Args: url: The remote file URL. Returns: The local temporary file path. """ suffix = "" match = re.search(r"\.([a-zA-Z0-9]{1,8})(?:\?|$)", url) if match: suffix = "." + match.group(1) fd, temp_path = tempfile.mkstemp(suffix=suffix) os.close(fd) r = requests.get(url, timeout=60) r.raise_for_status() with open(temp_path, "wb") as f: f.write(r.content) return temp_path @tool def inspect_local_text_file(path: str, max_chars: int = 12000) -> str: """ Read a local text-like file and return a preview. Args: path: The local file path. max_chars: Maximum number of characters to return. Returns: A text preview. """ p = Path(path) try: return p.read_text(encoding="utf-8", errors="ignore")[:max_chars] except Exception as e: return f"READ_ERROR: {e}" @tool def analyze_spreadsheet(path: str, instruction: str) -> str: """ Load a spreadsheet and return a compact JSON summary. Args: path: The local spreadsheet path. instruction: What the agent wants to determine. Returns: A JSON summary of the sheet. """ p = Path(path) if p.suffix.lower() == ".csv": df = pd.read_csv(path) elif p.suffix.lower() == ".tsv": df = pd.read_csv(path, sep="\t") else: df = pd.read_excel(path) payload = { "instruction": instruction, "shape": list(df.shape), "columns": list(df.columns), "dtypes": {c: str(t) for c, t in df.dtypes.items()}, "head": df.head(10).to_dict(orient="records"), "numeric_column_sums": { c: float(df[c].sum()) for c in df.select_dtypes(include="number").columns }, } return json.dumps(payload, ensure_ascii=False) @tool def python_compute(code: str) -> str: """ Execute short Python code for local calculations. Args: code: Python code that must assign the final value to a variable named result. Returns: The string form of result. """ local_vars = {} global_vars = { "pd": pd, "Path": Path, "json": json, "re": re, } exec(code, global_vars, local_vars) if "result" not in local_vars: return "ERROR_NO_RESULT" return str(local_vars["result"]) def clean_final_answer(text) -> str: if text is None: return "" answer = str(text).strip() answer = answer.replace("```", "").strip() answer = re.sub(r"^FINAL ANSWER\s*:\s*", "", answer, flags=re.IGNORECASE) answer = re.sub(r"^Answer\s*:\s*", "", answer, flags=re.IGNORECASE) answer = re.sub(r"[ \t]+", " ", answer).strip() if answer.startswith('"') and answer.endswith('"') and len(answer) >= 2: answer = answer[1:-1].strip() bad_markers = [ "incorrect api key", "openai_api_key", "error in generating model output", "traceback", "exception", "401", "403", "429", ] lowered = answer.lower() if any(marker in lowered for marker in bad_markers): return "" lines = [line.strip() for line in answer.splitlines() if line.strip()] if len(lines) > 1: answer = lines[-1] return answer.strip() SYSTEM_PROMPT = """ You are solving GAIA-style benchmark tasks. Rules: - Return only the final answer, with no explanation. - Respect the exact format requested in the question. - If a file is needed, use get_task_file_url(task_id), then download_file(url). - If the file is text-like, inspect it. - If it is a spreadsheet, analyze it. - Use web search when needed. - Double-check dates, numbers, names, and formatting. - Never output tool errors or debugging text as the final answer. - If you are uncertain, reason carefully and still output one best final answer only. """ class GaiaAgent: def __init__(self): api_key = os.environ.get("OPENAI_API_KEY") if not api_key: raise ValueError("OPENAI_API_KEY is missing") model_id = os.environ.get("OPENAI_MODEL", "gpt-4o-mini") self.model = OpenAIServerModel( model_id=model_id, api_base="https://api.openai.com/v1", api_key=api_key, ) try: self.agent = CodeAgent( model=self.model, tools=[ DuckDuckGoSearchTool(), VisitWebpageTool(), get_task_file_url, download_file, inspect_local_text_file, analyze_spreadsheet, python_compute, ], add_base_tools=False, max_steps=12, instructions=SYSTEM_PROMPT, ) except TypeError: self.agent = CodeAgent( model=self.model, tools=[ DuckDuckGoSearchTool(), VisitWebpageTool(), get_task_file_url, download_file, inspect_local_text_file, analyze_spreadsheet, python_compute, ], add_base_tools=False, max_steps=12, ) def answer_task(self, question: str, task_id: str, retries: int = 2) -> str: prompt = f""" Task ID: {task_id} Question: {question} Important: - Return only the final answer. - If a file is required, use get_task_file_url(task_id). """ last_error = None for attempt in range(retries + 1): try: raw = self.agent.run(prompt) cleaned = clean_final_answer(raw) if cleaned: return cleaned except Exception as e: last_error = str(e) if attempt < retries: time.sleep(1.5) if last_error: print(f"Task {task_id} failed: {last_error}") return "" def run_and_submit_all(profile: gr.OAuthProfile | None): space_id = os.getenv("SPACE_ID", "") if not profile: return "Please log in to Hugging Face first.", None try: agent = GaiaAgent() except Exception as e: return f"Error initializing agent: {e}", None try: r = requests.get(f"{API_URL}/questions", timeout=30) r.raise_for_status() questions = r.json() except Exception as e: return f"Error fetching questions: {e}", None answers = [] rows = [] for item in questions: task_id = item.get("task_id", "") question = item.get("question", "") if not task_id or not question: continue answer = agent.answer_task(question, task_id, retries=2) answers.append( { "task_id": task_id, "submitted_answer": answer, } ) rows.append( { "Task ID": task_id, "Question": question, "Submitted Answer": answer, } ) payload = { "username": profile.username.strip(), "agent_code": f"https://huggingface.co/spaces/{space_id}/tree/main" if space_id else "", "answers": answers, } try: r = requests.post(f"{API_URL}/submit", json=payload, timeout=180) r.raise_for_status() result = r.json() status = ( f"Submission successful.\n" f"User: {result.get('username')}\n" f"Score: {result.get('score', 'N/A')}%\n" f"Correct: {result.get('correct_count', '?')}/{result.get('total_attempted', '?')}\n" f"Message: {result.get('message', '')}" ) return status, pd.DataFrame(rows) except Exception as e: return f"Submission failed: {e}", pd.DataFrame(rows) with gr.Blocks() as demo: gr.Markdown("# Unit 4 GAIA Agent") gr.Markdown( """ 1. Add `OPENAI_API_KEY` as a Space secret 2. Optionally add `OPENAI_MODEL` 3. Log in with Hugging Face 4. Run evaluation """ ) gr.LoginButton() run_button = gr.Button("Run Evaluation and Submit") status_output = gr.Textbox(label="Status", lines=8, interactive=False) results_table = gr.DataFrame(label="Task Results", wrap=True) run_button.click( fn=run_and_submit_all, outputs=[status_output, results_table], ) if __name__ == "__main__": demo.launch(debug=True)