Files changed (1) hide show
  1. app.py +898 -121
app.py CHANGED
@@ -1,196 +1,973 @@
1
  import os
2
  import gradio as gr
3
  import requests
4
- import inspect
5
  import pandas as pd
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6
 
7
- # (Keep Constants as is)
8
- # --- Constants ---
9
  DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
10
 
11
- # --- Basic Agent Definition ---
12
- # ----- THIS IS WERE YOU CAN BUILD WHAT YOU WANT ------
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
13
  class BasicAgent:
 
14
  def __init__(self):
15
- print("BasicAgent initialized.")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
  def __call__(self, question: str) -> str:
17
- print(f"Agent received question (first 50 chars): {question[:50]}...")
18
- fixed_answer = "This is a default answer."
19
- print(f"Agent returning fixed answer: {fixed_answer}")
20
- return fixed_answer
21
 
22
- def run_and_submit_all( profile: gr.OAuthProfile | None):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
23
  """
24
- Fetches all questions, runs the BasicAgent on them, submits all answers,
25
- and displays the results.
 
26
  """
27
- # --- Determine HF Space Runtime URL and Repo URL ---
28
- space_id = os.getenv("SPACE_ID") # Get the SPACE_ID for sending link to the code
 
 
 
 
29
 
30
  if profile:
31
- username= f"{profile.username}"
32
- print(f"User logged in: {username}")
 
 
 
 
 
33
  else:
 
34
  print("User not logged in.")
35
- return "Please Login to Hugging Face with the button.", None
 
 
 
 
 
 
 
 
 
36
 
37
  api_url = DEFAULT_API_URL
 
38
  questions_url = f"{api_url}/questions"
 
39
  submit_url = f"{api_url}/submit"
40
 
41
- # 1. Instantiate Agent ( modify this part to create your agent)
 
 
 
 
42
  try:
 
43
  agent = BasicAgent()
 
44
  except Exception as e:
45
- print(f"Error instantiating agent: {e}")
46
- return f"Error initializing agent: {e}", None
47
- # In the case of an app running as a hugging Face space, this link points toward your codebase ( usefull for others so please keep it public)
48
- agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main"
49
- print(agent_code)
50
-
51
- # 2. Fetch Questions
52
- print(f"Fetching questions from: {questions_url}")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
53
  try:
54
- response = requests.get(questions_url, timeout=15)
 
 
 
 
 
55
  response.raise_for_status()
 
56
  questions_data = response.json()
 
57
  if not questions_data:
58
- print("Fetched questions list is empty.")
59
- return "Fetched questions list is empty or invalid format.", None
60
- print(f"Fetched {len(questions_data)} questions.")
61
- except requests.exceptions.RequestException as e:
62
- print(f"Error fetching questions: {e}")
63
- return f"Error fetching questions: {e}", None
64
- except requests.exceptions.JSONDecodeError as e:
65
- print(f"Error decoding JSON response from questions endpoint: {e}")
66
- print(f"Response text: {response.text[:500]}")
67
- return f"Error decoding server response for questions: {e}", None
68
  except Exception as e:
69
- print(f"An unexpected error occurred fetching questions: {e}")
70
- return f"An unexpected error occurred fetching questions: {e}", None
71
 
72
- # 3. Run your Agent
 
 
 
 
 
 
 
 
 
 
 
 
 
73
  results_log = []
 
74
  answers_payload = []
75
- print(f"Running agent on {len(questions_data)} questions...")
76
- for item in questions_data:
 
 
 
 
 
 
 
 
 
 
77
  task_id = item.get("task_id")
78
- question_text = item.get("question")
 
 
 
 
 
 
 
 
 
 
79
  if not task_id or question_text is None:
80
- print(f"Skipping item with missing task_id or question: {item}")
 
 
 
 
81
  continue
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
82
  try:
83
- submitted_answer = agent(question_text)
84
- answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer})
85
- results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer})
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
86
  except Exception as e:
87
- print(f"Error running agent on task {task_id}: {e}")
88
- results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"})
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
89
 
90
  if not answers_payload:
91
- print("Agent did not produce any answers to submit.")
92
- return "Agent did not produce any answers to submit.", pd.DataFrame(results_log)
93
 
94
- # 4. Prepare Submission
95
- submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload}
96
- status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..."
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
97
  print(status_update)
98
 
99
- # 5. Submit
100
- print(f"Submitting {len(answers_payload)} answers to: {submit_url}")
 
 
 
101
  try:
102
- response = requests.post(submit_url, json=submission_data, timeout=60)
 
 
 
 
 
 
 
 
 
103
  response.raise_for_status()
 
104
  result_data = response.json()
 
 
105
  final_status = (
106
- f"Submission Successful!\n"
107
- f"User: {result_data.get('username')}\n"
108
- f"Overall Score: {result_data.get('score', 'N/A')}% "
109
- f"({result_data.get('correct_count', '?')}/{result_data.get('total_attempted', '?')} correct)\n"
110
- f"Message: {result_data.get('message', 'No message received.')}"
111
- )
112
- print("Submission successful.")
113
- results_df = pd.DataFrame(results_log)
114
- return final_status, results_df
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
115
  except requests.exceptions.HTTPError as e:
116
- error_detail = f"Server responded with status {e.response.status_code}."
 
 
 
 
 
117
  try:
 
118
  error_json = e.response.json()
119
- error_detail += f" Detail: {error_json.get('detail', e.response.text)}"
120
- except requests.exceptions.JSONDecodeError:
121
- error_detail += f" Response: {e.response.text[:500]}"
122
- status_message = f"Submission Failed: {error_detail}"
123
- print(status_message)
124
- results_df = pd.DataFrame(results_log)
125
- return status_message, results_df
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
126
  except requests.exceptions.Timeout:
127
- status_message = "Submission Failed: The request timed out."
128
- print(status_message)
129
- results_df = pd.DataFrame(results_log)
130
- return status_message, results_df
 
 
 
 
 
 
 
 
131
  except requests.exceptions.RequestException as e:
132
- status_message = f"Submission Failed: Network error - {e}"
133
- print(status_message)
134
- results_df = pd.DataFrame(results_log)
135
- return status_message, results_df
 
 
 
 
 
 
 
 
136
  except Exception as e:
137
- status_message = f"An unexpected error occurred during submission: {e}"
138
- print(status_message)
139
- results_df = pd.DataFrame(results_log)
140
- return status_message, results_df
141
 
 
 
 
 
 
 
 
 
 
 
 
 
 
142
 
143
- # --- Build Gradio Interface using Blocks ---
144
  with gr.Blocks() as demo:
145
- gr.Markdown("# Basic Agent Evaluation Runner")
 
 
 
 
146
  gr.Markdown(
147
  """
148
- **Instructions:**
149
 
150
- 1. Please clone this space, then modify the code to define your agent's logic, the tools, the necessary packages, etc ...
151
- 2. Log in to your Hugging Face account using the button below. This uses your HF username for submission.
152
- 3. Click 'Run Evaluation & Submit All Answers' to fetch questions, run your agent, submit answers, and see the score.
 
 
153
 
154
- ---
155
- **Disclaimers:**
156
- Once clicking on the "submit button, it can take quite some time ( this is the time for the agent to go through all the questions).
157
- This space provides a basic setup and is intentionally sub-optimal to encourage you to develop your own, more robust solution. For instance for the delay process of the submit button, a solution could be to cache the answers and submit in a seperate action or even to answer the questions in async.
158
  """
159
  )
160
 
161
  gr.LoginButton()
162
 
163
- run_button = gr.Button("Run Evaluation & Submit All Answers")
 
 
 
 
 
 
 
 
 
 
 
 
 
164
 
165
- status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False)
166
- # Removed max_rows=10 from DataFrame constructor
167
- results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True)
168
 
169
  run_button.click(
 
170
  fn=run_and_submit_all,
171
- outputs=[status_output, results_table]
 
 
 
 
172
  )
173
 
 
 
 
 
 
174
  if __name__ == "__main__":
175
- print("\n" + "-"*30 + " App Starting " + "-"*30)
176
- # Check for SPACE_HOST and SPACE_ID at startup for information
177
- space_host_startup = os.getenv("SPACE_HOST")
178
- space_id_startup = os.getenv("SPACE_ID") # Get SPACE_ID at startup
179
-
180
- if space_host_startup:
181
- print(f"βœ… SPACE_HOST found: {space_host_startup}")
182
- print(f" Runtime URL should be: https://{space_host_startup}.hf.space")
183
- else:
184
- print("ℹ️ SPACE_HOST environment variable not found (running locally?).")
185
 
186
- if space_id_startup: # Print repo URLs if SPACE_ID is found
187
- print(f"βœ… SPACE_ID found: {space_id_startup}")
188
- print(f" Repo URL: https://huggingface.co/spaces/{space_id_startup}")
189
- print(f" Repo Tree URL: https://huggingface.co/spaces/{space_id_startup}/tree/main")
190
- else:
191
- print("ℹ️ SPACE_ID environment variable not found (running locally?). Repo URL cannot be determined.")
192
 
193
- print("-"*(60 + len(" App Starting ")) + "\n")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
194
 
195
- print("Launching Gradio Interface for Basic Agent Evaluation...")
196
- demo.launch(debug=True, share=False)
 
 
 
1
  import os
2
  import gradio as gr
3
  import requests
 
4
  import pandas as pd
5
+ import tempfile
6
+ import json
7
+ import subprocess
8
+ import sys
9
+ from pathlib import Path
10
+
11
+ from smolagents import (
12
+ CodeAgent,
13
+ HfApiModel,
14
+ DuckDuckGoSearchTool,
15
+ tool,
16
+ )
17
+
18
+ # ============================================================
19
+ # CONSTANTS
20
+ # ============================================================
21
 
 
 
22
  DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
23
 
24
+
25
+ # ============================================================
26
+ # CUSTOM TOOLS
27
+ # ============================================================
28
+
29
+ @tool
30
+ def calculate(expression: str) -> str:
31
+ """
32
+ Calculate a mathematical expression.
33
+
34
+ Args:
35
+ expression: A Python mathematical expression such as
36
+ "17 * 23" or "(100 / 4) + 7".
37
+
38
+ Returns:
39
+ The calculated result as a string.
40
+ """
41
+ try:
42
+ allowed = {
43
+ "abs": abs,
44
+ "round": round,
45
+ "min": min,
46
+ "max": max,
47
+ "sum": sum,
48
+ }
49
+
50
+ result = eval(
51
+ expression,
52
+ {"__builtins__": {}},
53
+ allowed
54
+ )
55
+
56
+ return str(result)
57
+
58
+ except Exception as e:
59
+ return f"Calculation error: {e}"
60
+
61
+
62
+ @tool
63
+ def analyze_csv_or_excel(file_path: str, operation: str) -> str:
64
+ """
65
+ Analyze a CSV or Excel file.
66
+
67
+ Args:
68
+ file_path: Path to the CSV or Excel file.
69
+ operation: Describe what should be calculated or inspected.
70
+
71
+ Returns:
72
+ A textual summary of the file and requested operation.
73
+ """
74
+
75
+ try:
76
+ path = Path(file_path)
77
+
78
+ if not path.exists():
79
+ return "File does not exist."
80
+
81
+ if path.suffix.lower() in [".xlsx", ".xls"]:
82
+ df = pd.read_excel(path)
83
+ elif path.suffix.lower() == ".csv":
84
+ df = pd.read_csv(path)
85
+ else:
86
+ return "Unsupported file type."
87
+
88
+ result = {
89
+ "columns": list(df.columns),
90
+ "rows": len(df),
91
+ "preview": df.head(10).to_dict(orient="records"),
92
+ "numeric_summary": df.describe(
93
+ include="all"
94
+ ).to_dict()
95
+ }
96
+
97
+ return json.dumps(result, default=str)
98
+
99
+ except Exception as e:
100
+ return f"Spreadsheet analysis error: {e}"
101
+
102
+
103
+ @tool
104
+ def read_text_file(file_path: str) -> str:
105
+ """
106
+ Read a text-based file.
107
+
108
+ Args:
109
+ file_path: Path to the text file.
110
+
111
+ Returns:
112
+ The contents of the file.
113
+ """
114
+
115
+ try:
116
+ path = Path(file_path)
117
+
118
+ if not path.exists():
119
+ return "File does not exist."
120
+
121
+ return path.read_text(
122
+ encoding="utf-8",
123
+ errors="ignore"
124
+ )
125
+
126
+ except Exception as e:
127
+ return f"Could not read file: {e}"
128
+
129
+
130
+ @tool
131
+ def execute_python_file(file_path: str) -> str:
132
+ """
133
+ Execute a Python file and return its output.
134
+
135
+ Args:
136
+ file_path: Path to the Python file.
137
+
138
+ Returns:
139
+ Standard output produced by the Python program.
140
+ """
141
+
142
+ try:
143
+ path = Path(file_path)
144
+
145
+ if not path.exists():
146
+ return "Python file does not exist."
147
+
148
+ result = subprocess.run(
149
+ [sys.executable, str(path)],
150
+ capture_output=True,
151
+ text=True,
152
+ timeout=30
153
+ )
154
+
155
+ output = result.stdout.strip()
156
+
157
+ if result.stderr:
158
+ output += "\nSTDERR:\n" + result.stderr.strip()
159
+
160
+ return output
161
+
162
+ except subprocess.TimeoutExpired:
163
+ return "Python execution timed out."
164
+
165
+ except Exception as e:
166
+ return f"Python execution error: {e}"
167
+
168
+
169
+ # ============================================================
170
+ # GAIA AGENT
171
+ # ============================================================
172
+
173
  class BasicAgent:
174
+
175
  def __init__(self):
176
+
177
+ print("=" * 60)
178
+ print("Initializing GAIA Agent...")
179
+ print("=" * 60)
180
+
181
+ # ----------------------------------------------------
182
+ # Hugging Face model
183
+ # ----------------------------------------------------
184
+
185
+ self.model = HfApiModel(
186
+ model_id="Qwen/Qwen2.5-Coder-32B-Instruct"
187
+ )
188
+
189
+ # ----------------------------------------------------
190
+ # Web search
191
+ # ----------------------------------------------------
192
+
193
+ self.search_tool = DuckDuckGoSearchTool()
194
+
195
+ # ----------------------------------------------------
196
+ # Agent
197
+ # ----------------------------------------------------
198
+
199
+ self.agent = CodeAgent(
200
+
201
+ model=self.model,
202
+
203
+ tools=[
204
+ self.search_tool,
205
+ calculate,
206
+ analyze_csv_or_excel,
207
+ read_text_file,
208
+ execute_python_file,
209
+ ],
210
+
211
+ additional_authorized_imports=[
212
+ "requests",
213
+ "bs4",
214
+ "datetime",
215
+ "pandas",
216
+ "math",
217
+ "statistics",
218
+ "json",
219
+ "re",
220
+ ],
221
+
222
+ max_steps=15,
223
+
224
+ verbosity_level=1,
225
+ )
226
+
227
+ print("GAIA Agent initialized successfully.")
228
+
229
+
230
+ # --------------------------------------------------------
231
+ # Clean final answer
232
+ # --------------------------------------------------------
233
+
234
+ def clean_answer(self, answer):
235
+
236
+ if answer is None:
237
+ return "unknown"
238
+
239
+ answer = str(answer).strip()
240
+
241
+ # Remove common unwanted prefixes
242
+ prefixes = [
243
+ "FINAL ANSWER:",
244
+ "Final Answer:",
245
+ "FINAL ANSWER",
246
+ "Final answer:",
247
+ "Answer:",
248
+ "answer:",
249
+ ]
250
+
251
+ for prefix in prefixes:
252
+
253
+ if answer.startswith(prefix):
254
+ answer = answer[len(prefix):].strip()
255
+
256
+ # Remove markdown code fences
257
+ answer = answer.replace("```", "").strip()
258
+
259
+ # Remove accidental surrounding quotes
260
+ if (
261
+ len(answer) >= 2
262
+ and answer[0] == '"'
263
+ and answer[-1] == '"'
264
+ ):
265
+ answer = answer[1:-1].strip()
266
+
267
+ return answer
268
+
269
+
270
+ # --------------------------------------------------------
271
+ # Run Agent
272
+ # --------------------------------------------------------
273
+
274
  def __call__(self, question: str) -> str:
 
 
 
 
275
 
276
+ print("\n" + "=" * 70)
277
+ print("QUESTION")
278
+ print(question)
279
+ print("=" * 70)
280
+
281
+ prompt = f"""
282
+ You are a highly capable general-purpose AI agent solving a
283
+ GAIA Level-1 benchmark question.
284
+
285
+ Your goal is to obtain the CORRECT final answer.
286
+
287
+ You have access to:
288
+
289
+ 1. Web search
290
+ 2. Python calculations
291
+ 3. Spreadsheet analysis
292
+ 4. Python file execution
293
+ 5. Text file reading
294
+
295
+ GENERAL RULES:
296
+
297
+ - Carefully understand the question.
298
+ - Do not guess if you can verify the information.
299
+ - Use web search for factual/current/research questions.
300
+ - Use Python for calculations, sorting, counting and data processing.
301
+ - Use spreadsheet tools for Excel/CSV questions.
302
+ - Use the Python execution tool when a Python file is provided.
303
+ - Perform multi-step reasoning when necessary.
304
+ - Verify important results before answering.
305
+
306
+ IMPORTANT OUTPUT RULE:
307
+
308
+ The benchmark uses exact matching.
309
+
310
+ Therefore your FINAL RESPONSE must contain ONLY the answer.
311
+
312
+ Do NOT write:
313
+
314
+ "FINAL ANSWER:"
315
+
316
+ Do NOT write:
317
+
318
+ "The answer is..."
319
+
320
+ Do NOT explain your reasoning.
321
+
322
+ Do NOT add unnecessary punctuation.
323
+
324
+ Examples:
325
+
326
+ If the answer is a number:
327
+
328
+ 391
329
+
330
+ If the answer is a name:
331
+
332
+ Smith
333
+
334
+ If the question asks for a comma-separated list:
335
+
336
+ apples, bananas, oranges
337
+
338
+ Question:
339
+
340
+ {question}
341
+ """
342
+
343
+ try:
344
+
345
+ result = self.agent.run(prompt)
346
+
347
+ answer = self.clean_answer(result)
348
+
349
+ print("\nAGENT ANSWER:")
350
+ print(answer)
351
+
352
+ return answer
353
+
354
+ except Exception as e:
355
+
356
+ print("\nAGENT ERROR:")
357
+ print(e)
358
+
359
+ return "unknown"
360
+
361
+
362
+ # ============================================================
363
+ # DOWNLOAD ATTACHED GAIA FILE
364
+ # ============================================================
365
+
366
+ def download_task_file(task_id):
367
+
368
+ file_url = f"{DEFAULT_API_URL}/files/{task_id}"
369
+
370
+ try:
371
+
372
+ response = requests.get(
373
+ file_url,
374
+ timeout=30
375
+ )
376
+
377
+ if response.status_code != 200:
378
+ return None
379
+
380
+ # Try to determine extension from headers
381
+ content_type = response.headers.get(
382
+ "content-type",
383
+ ""
384
+ ).lower()
385
+
386
+ extension = ".bin"
387
+
388
+ if "png" in content_type:
389
+ extension = ".png"
390
+
391
+ elif "jpeg" in content_type:
392
+ extension = ".jpg"
393
+
394
+ elif "audio" in content_type:
395
+ extension = ".mp3"
396
+
397
+ elif "excel" in content_type:
398
+ extension = ".xlsx"
399
+
400
+ elif "csv" in content_type:
401
+ extension = ".csv"
402
+
403
+ elif "python" in content_type:
404
+ extension = ".py"
405
+
406
+ # Temporary file
407
+ temp = tempfile.NamedTemporaryFile(
408
+ delete=False,
409
+ suffix=extension
410
+ )
411
+
412
+ temp.write(response.content)
413
+ temp.close()
414
+
415
+ print(
416
+ f"Downloaded task file: {temp.name}"
417
+ )
418
+
419
+ return temp.name
420
+
421
+ except Exception as e:
422
+
423
+ print(
424
+ f"Could not download task file: {e}"
425
+ )
426
+
427
+ return None
428
+
429
+
430
+ # ============================================================
431
+ # MAIN EVALUATION FUNCTION
432
+ # ============================================================
433
+
434
+ def run_and_submit_all(
435
+ profile: gr.OAuthProfile | None
436
+ ):
437
+
438
  """
439
+ Fetch all GAIA questions,
440
+ run the Agent,
441
+ and submit answers.
442
  """
443
+
444
+ # --------------------------------------------------------
445
+ # Get username
446
+ # --------------------------------------------------------
447
+
448
+ space_id = os.getenv("SPACE_ID")
449
 
450
  if profile:
451
+
452
+ username = f"{profile.username}"
453
+
454
+ print(
455
+ f"User logged in: {username}"
456
+ )
457
+
458
  else:
459
+
460
  print("User not logged in.")
461
+
462
+ return (
463
+ "Please Login to Hugging Face with the button.",
464
+ None
465
+ )
466
+
467
+
468
+ # --------------------------------------------------------
469
+ # API endpoints
470
+ # --------------------------------------------------------
471
 
472
  api_url = DEFAULT_API_URL
473
+
474
  questions_url = f"{api_url}/questions"
475
+
476
  submit_url = f"{api_url}/submit"
477
 
478
+
479
+ # --------------------------------------------------------
480
+ # Create Agent
481
+ # --------------------------------------------------------
482
+
483
  try:
484
+
485
  agent = BasicAgent()
486
+
487
  except Exception as e:
488
+
489
+ print(
490
+ f"Error initializing agent: {e}"
491
+ )
492
+
493
+ return (
494
+ f"Error initializing agent: {e}",
495
+ None
496
+ )
497
+
498
+
499
+ # --------------------------------------------------------
500
+ # Code link
501
+ # --------------------------------------------------------
502
+
503
+ agent_code = (
504
+ f"https://huggingface.co/spaces/"
505
+ f"{space_id}/tree/main"
506
+ )
507
+
508
+ print(
509
+ f"Agent code: {agent_code}"
510
+ )
511
+
512
+
513
+ # --------------------------------------------------------
514
+ # Fetch questions
515
+ # --------------------------------------------------------
516
+
517
+ print(
518
+ f"Fetching questions from: {questions_url}"
519
+ )
520
+
521
  try:
522
+
523
+ response = requests.get(
524
+ questions_url,
525
+ timeout=30
526
+ )
527
+
528
  response.raise_for_status()
529
+
530
  questions_data = response.json()
531
+
532
  if not questions_data:
533
+
534
+ return (
535
+ "Question list is empty.",
536
+ None
537
+ )
538
+
539
+ print(
540
+ f"Fetched {len(questions_data)} questions."
541
+ )
542
+
543
  except Exception as e:
 
 
544
 
545
+ print(
546
+ f"Error fetching questions: {e}"
547
+ )
548
+
549
+ return (
550
+ f"Error fetching questions: {e}",
551
+ None
552
+ )
553
+
554
+
555
+ # --------------------------------------------------------
556
+ # Run Agent
557
+ # --------------------------------------------------------
558
+
559
  results_log = []
560
+
561
  answers_payload = []
562
+
563
+ print(
564
+ f"\nRunning Agent on "
565
+ f"{len(questions_data)} questions..."
566
+ )
567
+
568
+
569
+ for index, item in enumerate(
570
+ questions_data,
571
+ start=1
572
+ ):
573
+
574
  task_id = item.get("task_id")
575
+
576
+ question_text = item.get(
577
+ "question"
578
+ )
579
+
580
+ file_name = item.get(
581
+ "file_name",
582
+ ""
583
+ )
584
+
585
+
586
  if not task_id or question_text is None:
587
+
588
+ print(
589
+ "Skipping invalid question."
590
+ )
591
+
592
  continue
593
+
594
+
595
+ print("\n")
596
+ print("#" * 80)
597
+ print(
598
+ f"QUESTION {index}/{len(questions_data)}"
599
+ )
600
+ print(
601
+ f"TASK ID: {task_id}"
602
+ )
603
+
604
+ if file_name:
605
+
606
+ print(
607
+ f"ATTACHED FILE: {file_name}"
608
+ )
609
+
610
+
611
+ # ----------------------------------------------------
612
+ # Handle attached file
613
+ # ----------------------------------------------------
614
+
615
+ enhanced_question = question_text
616
+
617
+ if file_name:
618
+
619
+ file_path = download_task_file(
620
+ task_id
621
+ )
622
+
623
+ if file_path:
624
+
625
+ enhanced_question += f"""
626
+
627
+ IMPORTANT:
628
+
629
+ This GAIA question has an attached file.
630
+
631
+ Original file name:
632
+ {file_name}
633
+
634
+ The file has been downloaded locally to:
635
+
636
+ {file_path}
637
+
638
+ You MUST use the appropriate tool to inspect this file
639
+ when the question requires information from it.
640
+
641
+ For an Excel/CSV file:
642
+ use analyze_csv_or_excel.
643
+
644
+ For a Python file:
645
+ use execute_python_file.
646
+
647
+ For a text file:
648
+ use read_text_file.
649
+
650
+ Do not ignore the attached file.
651
+ """
652
+
653
+ else:
654
+
655
+ enhanced_question += f"""
656
+
657
+ IMPORTANT:
658
+
659
+ The question references this attached file:
660
+
661
+ {file_name}
662
+
663
+ The file could not be downloaded automatically.
664
+ Try to solve the question using the information available
665
+ in the question or other available tools.
666
+ """
667
+
668
+
669
+ # ----------------------------------------------------
670
+ # Run Agent
671
+ # ----------------------------------------------------
672
+
673
  try:
674
+
675
+ submitted_answer = agent(
676
+ enhanced_question
677
+ )
678
+
679
+ submitted_answer = str(
680
+ submitted_answer
681
+ ).strip()
682
+
683
+ answers_payload.append(
684
+ {
685
+ "task_id": task_id,
686
+ "submitted_answer": submitted_answer
687
+ }
688
+ )
689
+
690
+ results_log.append(
691
+ {
692
+ "Task ID": task_id,
693
+ "Question": question_text,
694
+ "File": file_name,
695
+ "Submitted Answer": submitted_answer
696
+ }
697
+ )
698
+
699
+ print(
700
+ f"FINAL ANSWER: {submitted_answer}"
701
+ )
702
+
703
+
704
  except Exception as e:
705
+
706
+ print(
707
+ f"Agent failed on task "
708
+ f"{task_id}: {e}"
709
+ )
710
+
711
+ results_log.append(
712
+ {
713
+ "Task ID": task_id,
714
+ "Question": question_text,
715
+ "File": file_name,
716
+ "Submitted Answer":
717
+ f"AGENT ERROR: {e}"
718
+ }
719
+ )
720
+
721
+
722
+ # --------------------------------------------------------
723
+ # Check answers
724
+ # --------------------------------------------------------
725
 
726
  if not answers_payload:
 
 
727
 
728
+ return (
729
+ "Agent did not produce any answers.",
730
+ pd.DataFrame(results_log)
731
+ )
732
+
733
+
734
+ # --------------------------------------------------------
735
+ # Prepare submission
736
+ # --------------------------------------------------------
737
+
738
+ submission_data = {
739
+
740
+ "username":
741
+ username.strip(),
742
+
743
+ "agent_code":
744
+ agent_code,
745
+
746
+ "answers":
747
+ answers_payload
748
+ }
749
+
750
+
751
+ status_update = (
752
+ f"Agent finished.\n"
753
+ f"Submitting {len(answers_payload)} "
754
+ f"answers for user '{username}'..."
755
+ )
756
+
757
  print(status_update)
758
 
759
+
760
+ # --------------------------------------------------------
761
+ # Submit
762
+ # --------------------------------------------------------
763
+
764
  try:
765
+
766
+ response = requests.post(
767
+
768
+ submit_url,
769
+
770
+ json=submission_data,
771
+
772
+ timeout=180
773
+ )
774
+
775
  response.raise_for_status()
776
+
777
  result_data = response.json()
778
+
779
+
780
  final_status = (
781
+
782
+ "Submission Successful!\n\n"
783
+
784
+ f"User: "
785
+ f"{result_data.get('username')}\n"
786
+
787
+ f"Overall Score: "
788
+ f"{result_data.get('score', 'N/A')}%\n"
789
+
790
+ f"Correct: "
791
+ f"{result_data.get('correct_count', '?')}/"
792
+ f"{result_data.get('total_attempted', '?')}\n\n"
793
+
794
+ f"Message: "
795
+ f"{result_data.get('message', '')}"
796
+ )
797
+
798
+
799
+ print(final_status)
800
+
801
+
802
+ results_df = pd.DataFrame(
803
+ results_log
804
+ )
805
+
806
+
807
+ return (
808
+ final_status,
809
+ results_df
810
+ )
811
+
812
+
813
  except requests.exceptions.HTTPError as e:
814
+
815
+ error_detail = (
816
+ f"Server responded with "
817
+ f"status {e.response.status_code}."
818
+ )
819
+
820
  try:
821
+
822
  error_json = e.response.json()
823
+
824
+ error_detail += (
825
+ f" Detail: "
826
+ f"{error_json.get('detail', '')}"
827
+ )
828
+
829
+ except Exception:
830
+
831
+ error_detail += (
832
+ f" Response: "
833
+ f"{e.response.text[:500]}"
834
+ )
835
+
836
+
837
+ status_message = (
838
+ f"Submission Failed: "
839
+ f"{error_detail}"
840
+ )
841
+
842
+
843
+ return (
844
+ status_message,
845
+ pd.DataFrame(results_log)
846
+ )
847
+
848
+
849
  except requests.exceptions.Timeout:
850
+
851
+ status_message = (
852
+ "Submission Failed: "
853
+ "The request timed out."
854
+ )
855
+
856
+ return (
857
+ status_message,
858
+ pd.DataFrame(results_log)
859
+ )
860
+
861
+
862
  except requests.exceptions.RequestException as e:
863
+
864
+ status_message = (
865
+ f"Submission Failed: "
866
+ f"Network error - {e}"
867
+ )
868
+
869
+ return (
870
+ status_message,
871
+ pd.DataFrame(results_log)
872
+ )
873
+
874
+
875
  except Exception as e:
 
 
 
 
876
 
877
+ status_message = (
878
+ f"Unexpected submission error: {e}"
879
+ )
880
+
881
+ return (
882
+ status_message,
883
+ pd.DataFrame(results_log)
884
+ )
885
+
886
+
887
+ # ============================================================
888
+ # GRADIO INTERFACE
889
+ # ============================================================
890
 
 
891
  with gr.Blocks() as demo:
892
+
893
+ gr.Markdown(
894
+ "# πŸ€– GAIA Agent Evaluation Runner"
895
+ )
896
+
897
  gr.Markdown(
898
  """
899
+ ## Instructions
900
 
901
+ 1. Log in to your Hugging Face account.
902
+ 2. Your Agent will receive the GAIA evaluation questions.
903
+ 3. Your Agent will use tools to solve them.
904
+ 4. The answers will automatically be submitted.
905
+ 5. Your score will be displayed below.
906
 
907
+ **Important:** The benchmark uses exact matching,
908
+ so the Agent should return only the final answer.
 
 
909
  """
910
  )
911
 
912
  gr.LoginButton()
913
 
914
+ run_button = gr.Button(
915
+ "πŸš€ Run Evaluation & Submit All Answers"
916
+ )
917
+
918
+ status_output = gr.Textbox(
919
+ label="Run Status / Submission Result",
920
+ lines=7,
921
+ interactive=False
922
+ )
923
+
924
+ results_table = gr.DataFrame(
925
+ label="Questions and Agent Answers",
926
+ wrap=True
927
+ )
928
 
 
 
 
929
 
930
  run_button.click(
931
+
932
  fn=run_and_submit_all,
933
+
934
+ outputs=[
935
+ status_output,
936
+ results_table
937
+ ]
938
  )
939
 
940
+
941
+ # ============================================================
942
+ # START
943
+ # ============================================================
944
+
945
  if __name__ == "__main__":
 
 
 
 
 
 
 
 
 
 
946
 
947
+ print(
948
+ "\n" +
949
+ "=" * 60 +
950
+ "\n GAIA AGENT STARTING \n" +
951
+ "=" * 60
952
+ )
953
 
954
+ space_id = os.getenv(
955
+ "SPACE_ID"
956
+ )
957
+
958
+ if space_id:
959
+
960
+ print(
961
+ f"Space ID: {space_id}"
962
+ )
963
+
964
+ print(
965
+ "Code URL: "
966
+ f"https://huggingface.co/spaces/"
967
+ f"{space_id}/tree/main"
968
+ )
969
 
970
+ demo.launch(
971
+ debug=True,
972
+ share=False
973
+ )