PaulJ00's picture
Update app.py
6fc8489 verified
Raw History Blame Contribute Delete
9.36 kB
import os
import re
import json
import time
import tempfile
from pathlib import Path
import gradio as gr
import pandas as pd
import requests
from smolagents import CodeAgent, tool, DuckDuckGoSearchTool, VisitWebpageTool
try:
from smolagents import OpenAIServerModel
except ImportError:
from smolagents import OpenAIModel as OpenAIServerModel
API_URL = "https://agents-course-unit4-scoring.hf.space"
@tool
def get_task_file_url(task_id: str) -> str:
"""
Build the file URL for a GAIA task.
Args:
task_id: The task identifier.
Returns:
The download URL for the task file.
"""
return f"{API_URL}/files/{task_id}"
@tool
def download_file(url: str) -> str:
"""
Download a remote file to a temporary local path.
Args:
url: The remote file URL.
Returns:
The local temporary file path.
"""
suffix = ""
match = re.search(r"\.([a-zA-Z0-9]{1,8})(?:\?|$)", url)
if match:
suffix = "." + match.group(1)
fd, temp_path = tempfile.mkstemp(suffix=suffix)
os.close(fd)
r = requests.get(url, timeout=60)
r.raise_for_status()
with open(temp_path, "wb") as f:
f.write(r.content)
return temp_path
@tool
def inspect_local_text_file(path: str, max_chars: int = 12000) -> str:
"""
Read a local text-like file and return a preview.
Args:
path: The local file path.
max_chars: Maximum number of characters to return.
Returns:
A text preview.
"""
p = Path(path)
try:
return p.read_text(encoding="utf-8", errors="ignore")[:max_chars]
except Exception as e:
return f"READ_ERROR: {e}"
@tool
def analyze_spreadsheet(path: str, instruction: str) -> str:
"""
Load a spreadsheet and return a compact JSON summary.
Args:
path: The local spreadsheet path.
instruction: What the agent wants to determine.
Returns:
A JSON summary of the sheet.
"""
p = Path(path)
if p.suffix.lower() == ".csv":
df = pd.read_csv(path)
elif p.suffix.lower() == ".tsv":
df = pd.read_csv(path, sep="\t")
else:
df = pd.read_excel(path)
payload = {
"instruction": instruction,
"shape": list(df.shape),
"columns": list(df.columns),
"dtypes": {c: str(t) for c, t in df.dtypes.items()},
"head": df.head(10).to_dict(orient="records"),
"numeric_column_sums": {
c: float(df[c].sum())
for c in df.select_dtypes(include="number").columns
},
}
return json.dumps(payload, ensure_ascii=False)
@tool
def python_compute(code: str) -> str:
"""
Execute short Python code for local calculations.
Args:
code: Python code that must assign the final value to a variable named result.
Returns:
The string form of result.
"""
local_vars = {}
global_vars = {
"pd": pd,
"Path": Path,
"json": json,
"re": re,
}
exec(code, global_vars, local_vars)
if "result" not in local_vars:
return "ERROR_NO_RESULT"
return str(local_vars["result"])
def clean_final_answer(text) -> str:
if text is None:
return ""
answer = str(text).strip()
answer = answer.replace("```", "").strip()
answer = re.sub(r"^FINAL ANSWER\s*:\s*", "", answer, flags=re.IGNORECASE)
answer = re.sub(r"^Answer\s*:\s*", "", answer, flags=re.IGNORECASE)
answer = re.sub(r"[ \t]+", " ", answer).strip()
if answer.startswith('"') and answer.endswith('"') and len(answer) >= 2:
answer = answer[1:-1].strip()
bad_markers = [
"incorrect api key",
"openai_api_key",
"error in generating model output",
"traceback",
"exception",
"401",
"403",
"429",
]
lowered = answer.lower()
if any(marker in lowered for marker in bad_markers):
return ""
lines = [line.strip() for line in answer.splitlines() if line.strip()]
if len(lines) > 1:
answer = lines[-1]
return answer.strip()
SYSTEM_PROMPT = """
You are solving GAIA-style benchmark tasks.
Rules:
- Return only the final answer, with no explanation.
- Respect the exact format requested in the question.
- If a file is needed, use get_task_file_url(task_id), then download_file(url).
- If the file is text-like, inspect it.
- If it is a spreadsheet, analyze it.
- Use web search when needed.
- Double-check dates, numbers, names, and formatting.
- Never output tool errors or debugging text as the final answer.
- If you are uncertain, reason carefully and still output one best final answer only.
"""
class GaiaAgent:
def __init__(self):
api_key = os.environ.get("OPENAI_API_KEY")
if not api_key:
raise ValueError("OPENAI_API_KEY is missing")
model_id = os.environ.get("OPENAI_MODEL", "gpt-4o-mini")
self.model = OpenAIServerModel(
model_id=model_id,
api_base="https://api.openai.com/v1",
api_key=api_key,
)
try:
self.agent = CodeAgent(
model=self.model,
tools=[
DuckDuckGoSearchTool(),
VisitWebpageTool(),
get_task_file_url,
download_file,
inspect_local_text_file,
analyze_spreadsheet,
python_compute,
],
add_base_tools=False,
max_steps=12,
instructions=SYSTEM_PROMPT,
)
except TypeError:
self.agent = CodeAgent(
model=self.model,
tools=[
DuckDuckGoSearchTool(),
VisitWebpageTool(),
get_task_file_url,
download_file,
inspect_local_text_file,
analyze_spreadsheet,
python_compute,
],
add_base_tools=False,
max_steps=12,
)
def answer_task(self, question: str, task_id: str, retries: int = 2) -> str:
prompt = f"""
Task ID: {task_id}
Question:
{question}
Important:
- Return only the final answer.
- If a file is required, use get_task_file_url(task_id).
"""
last_error = None
for attempt in range(retries + 1):
try:
raw = self.agent.run(prompt)
cleaned = clean_final_answer(raw)
if cleaned:
return cleaned
except Exception as e:
last_error = str(e)
if attempt < retries:
time.sleep(1.5)
if last_error:
print(f"Task {task_id} failed: {last_error}")
return ""
def run_and_submit_all(profile: gr.OAuthProfile | None):
space_id = os.getenv("SPACE_ID", "")
if not profile:
return "Please log in to Hugging Face first.", None
try:
agent = GaiaAgent()
except Exception as e:
return f"Error initializing agent: {e}", None
try:
r = requests.get(f"{API_URL}/questions", timeout=30)
r.raise_for_status()
questions = r.json()
except Exception as e:
return f"Error fetching questions: {e}", None
answers = []
rows = []
for item in questions:
task_id = item.get("task_id", "")
question = item.get("question", "")
if not task_id or not question:
continue
answer = agent.answer_task(question, task_id, retries=2)
answers.append(
{
"task_id": task_id,
"submitted_answer": answer,
}
)
rows.append(
{
"Task ID": task_id,
"Question": question,
"Submitted Answer": answer,
}
)
payload = {
"username": profile.username.strip(),
"agent_code": f"https://huggingface.co/spaces/{space_id}/tree/main" if space_id else "",
"answers": answers,
}
try:
r = requests.post(f"{API_URL}/submit", json=payload, timeout=180)
r.raise_for_status()
result = r.json()
status = (
f"Submission successful.\n"
f"User: {result.get('username')}\n"
f"Score: {result.get('score', 'N/A')}%\n"
f"Correct: {result.get('correct_count', '?')}/{result.get('total_attempted', '?')}\n"
f"Message: {result.get('message', '')}"
)
return status, pd.DataFrame(rows)
except Exception as e:
return f"Submission failed: {e}", pd.DataFrame(rows)
with gr.Blocks() as demo:
gr.Markdown("# Unit 4 GAIA Agent")
gr.Markdown(
"""
1. Add `OPENAI_API_KEY` as a Space secret
2. Optionally add `OPENAI_MODEL`
3. Log in with Hugging Face
4. Run evaluation
"""
)
gr.LoginButton()
run_button = gr.Button("Run Evaluation and Submit")
status_output = gr.Textbox(label="Status", lines=8, interactive=False)
results_table = gr.DataFrame(label="Task Results", wrap=True)
run_button.click(
fn=run_and_submit_all,
outputs=[status_output, results_table],
)
if __name__ == "__main__":
demo.launch(debug=True)