Files changed (1) hide show
  1. app.py +98 -46
app.py CHANGED
@@ -1,34 +1,98 @@
1
  import os
2
  import gradio as gr
3
  import requests
4
- import inspect
5
  import pandas as pd
 
 
6
 
7
- # (Keep Constants as is)
8
  # --- Constants ---
9
  DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
10
 
11
- # --- Basic Agent Definition ---
12
- # ----- THIS IS WERE YOU CAN BUILD WHAT YOU WANT ------
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
13
  class BasicAgent:
14
  def __init__(self):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
15
  print("BasicAgent initialized.")
16
- def __call__(self, question: str) -> str:
 
17
  print(f"Agent received question (first 50 chars): {question[:50]}...")
18
- fixed_answer = "This is a default answer."
19
- print(f"Agent returning fixed answer: {fixed_answer}")
20
- return fixed_answer
21
 
22
- def run_and_submit_all( profile: gr.OAuthProfile | None):
23
- """
24
- Fetches all questions, runs the BasicAgent on them, submits all answers,
25
- and displays the results.
26
- """
27
- # --- Determine HF Space Runtime URL and Repo URL ---
28
- space_id = os.getenv("SPACE_ID") # Get the SPACE_ID for sending link to the code
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
29
 
30
  if profile:
31
- username= f"{profile.username}"
32
  print(f"User logged in: {username}")
33
  else:
34
  print("User not logged in.")
@@ -38,41 +102,38 @@ def run_and_submit_all( profile: gr.OAuthProfile | None):
38
  questions_url = f"{api_url}/questions"
39
  submit_url = f"{api_url}/submit"
40
 
41
- # 1. Instantiate Agent ( modify this part to create your agent)
42
  try:
43
  agent = BasicAgent()
44
  except Exception as e:
45
  print(f"Error instantiating agent: {e}")
46
  return f"Error initializing agent: {e}", None
47
- # In the case of an app running as a hugging Face space, this link points toward your codebase ( usefull for others so please keep it public)
48
  agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main"
49
  print(agent_code)
50
 
51
- # 2. Fetch Questions
52
  print(f"Fetching questions from: {questions_url}")
53
  try:
54
  response = requests.get(questions_url, timeout=15)
55
  response.raise_for_status()
56
  questions_data = response.json()
57
  if not questions_data:
58
- print("Fetched questions list is empty.")
59
- return "Fetched questions list is empty or invalid format.", None
60
  print(f"Fetched {len(questions_data)} questions.")
61
  except requests.exceptions.RequestException as e:
62
  print(f"Error fetching questions: {e}")
63
  return f"Error fetching questions: {e}", None
64
  except requests.exceptions.JSONDecodeError as e:
65
- print(f"Error decoding JSON response from questions endpoint: {e}")
66
- print(f"Response text: {response.text[:500]}")
67
- return f"Error decoding server response for questions: {e}", None
68
  except Exception as e:
69
  print(f"An unexpected error occurred fetching questions: {e}")
70
  return f"An unexpected error occurred fetching questions: {e}", None
71
 
72
- # 3. Run your Agent
73
  results_log = []
74
  answers_payload = []
75
  print(f"Running agent on {len(questions_data)} questions...")
 
76
  for item in questions_data:
77
  task_id = item.get("task_id")
78
  question_text = item.get("question")
@@ -80,23 +141,21 @@ def run_and_submit_all( profile: gr.OAuthProfile | None):
80
  print(f"Skipping item with missing task_id or question: {item}")
81
  continue
82
  try:
83
- submitted_answer = agent(question_text)
84
  answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer})
85
  results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer})
86
  except Exception as e:
87
- print(f"Error running agent on task {task_id}: {e}")
88
- results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"})
89
 
90
  if not answers_payload:
91
  print("Agent did not produce any answers to submit.")
92
  return "Agent did not produce any answers to submit.", pd.DataFrame(results_log)
93
 
94
- # 4. Prepare Submission
95
  submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload}
96
  status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..."
97
  print(status_update)
98
 
99
- # 5. Submit
100
  print(f"Submitting {len(answers_payload)} answers to: {submit_url}")
101
  try:
102
  response = requests.post(submit_url, json=submission_data, timeout=60)
@@ -140,21 +199,19 @@ def run_and_submit_all( profile: gr.OAuthProfile | None):
140
  return status_message, results_df
141
 
142
 
143
- # --- Build Gradio Interface using Blocks ---
144
  with gr.Blocks() as demo:
145
  gr.Markdown("# Basic Agent Evaluation Runner")
146
  gr.Markdown(
147
  """
148
  **Instructions:**
149
-
150
  1. Please clone this space, then modify the code to define your agent's logic, the tools, the necessary packages, etc ...
151
  2. Log in to your Hugging Face account using the button below. This uses your HF username for submission.
152
  3. Click 'Run Evaluation & Submit All Answers' to fetch questions, run your agent, submit answers, and see the score.
153
-
154
  ---
155
  **Disclaimers:**
156
  Once clicking on the "submit button, it can take quite some time ( this is the time for the agent to go through all the questions).
157
- This space provides a basic setup and is intentionally sub-optimal to encourage you to develop your own, more robust solution. For instance for the delay process of the submit button, a solution could be to cache the answers and submit in a seperate action or even to answer the questions in async.
158
  """
159
  )
160
 
@@ -163,7 +220,6 @@ with gr.Blocks() as demo:
163
  run_button = gr.Button("Run Evaluation & Submit All Answers")
164
 
165
  status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False)
166
- # Removed max_rows=10 from DataFrame constructor
167
  results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True)
168
 
169
  run_button.click(
@@ -173,24 +229,20 @@ with gr.Blocks() as demo:
173
 
174
  if __name__ == "__main__":
175
  print("\n" + "-"*30 + " App Starting " + "-"*30)
176
- # Check for SPACE_HOST and SPACE_ID at startup for information
177
  space_host_startup = os.getenv("SPACE_HOST")
178
- space_id_startup = os.getenv("SPACE_ID") # Get SPACE_ID at startup
179
 
180
  if space_host_startup:
181
- print(f"SPACE_HOST found: {space_host_startup}")
182
- print(f" Runtime URL should be: https://{space_host_startup}.hf.space")
183
  else:
184
- print("ℹ️ SPACE_HOST environment variable not found (running locally?).")
185
 
186
- if space_id_startup: # Print repo URLs if SPACE_ID is found
187
- print(f"SPACE_ID found: {space_id_startup}")
188
- print(f" Repo URL: https://huggingface.co/spaces/{space_id_startup}")
189
- print(f" Repo Tree URL: https://huggingface.co/spaces/{space_id_startup}/tree/main")
190
  else:
191
- print("ℹ️ SPACE_ID environment variable not found (running locally?). Repo URL cannot be determined.")
192
 
193
  print("-"*(60 + len(" App Starting ")) + "\n")
194
-
195
  print("Launching Gradio Interface for Basic Agent Evaluation...")
196
  demo.launch(debug=True, share=False)
 
1
  import os
2
  import gradio as gr
3
  import requests
 
4
  import pandas as pd
5
+ import re
6
+ from smolagents import CodeAgent, InferenceClientModel, DuckDuckGoSearchTool, VisitWebpageTool, tool
7
 
 
8
  # --- Constants ---
9
  DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
10
 
11
+ # --- Tools ---
12
+ @tool
13
+ def calculator(expression: str) -> str:
14
+ """Evaluates a mathematical expression and returns the result as a string.
15
+
16
+ Args:
17
+ expression: A valid Python math expression e.g. '847 * 293'.
18
+ """
19
+ try:
20
+ result = eval(expression, {"__builtins__": {}}, {})
21
+ return str(result)
22
+ except Exception as e:
23
+ return f"Error: {e}"
24
+
25
+
26
+ @tool
27
+ def fetch_task_file(task_id: str) -> str:
28
+ """Fetches the file associated with a GAIA task by its task ID.
29
+
30
+ Args:
31
+ task_id: The GAIA task ID string.
32
+ """
33
+ try:
34
+ url = f"{DEFAULT_API_URL}/files/{task_id}"
35
+ resp = requests.get(url, timeout=15)
36
+ resp.raise_for_status()
37
+ return resp.text
38
+ except Exception as e:
39
+ return f"FETCH_ERROR: {e}"
40
+
41
+
42
+ # --- Agent ---
43
  class BasicAgent:
44
  def __init__(self):
45
+ model = InferenceClientModel(
46
+ model_id="meta-llama/Llama-3.3-70B-Instruct",
47
+ timeout=120,
48
+ )
49
+ self.agent = CodeAgent(
50
+ tools=[
51
+ DuckDuckGoSearchTool(),
52
+ VisitWebpageTool(),
53
+ calculator,
54
+ fetch_task_file,
55
+ ],
56
+ model=model,
57
+ max_steps=5,
58
+ additional_authorized_imports=["json", "re", "datetime", "math"],
59
+ )
60
  print("BasicAgent initialized.")
61
+
62
+ def __call__(self, question: str, task_id: str = "") -> str:
63
  print(f"Agent received question (first 50 chars): {question[:50]}...")
 
 
 
64
 
65
+ prompt = f"""
66
+ Question: {question}
67
+ {"Use fetch_task_file(task_id='" + task_id + "') to retrieve the associated file if needed." if task_id else ""}
68
+
69
+ Rules:
70
+ - Use tools to find the correct answer.
71
+ - Return ONLY the final answer no explanation, no preamble, no punctuation unless required by the question.
72
+ - If the answer is a number, return only the number.
73
+ - If the answer is a name, return only the name.
74
+ - If the answer is a list, return items separated by commas.
75
+ - Never include the words FINAL ANSWER in your response.
76
+ """
77
+ try:
78
+ result = self.agent.run(prompt)
79
+ # Strip common verbose prefixes the model might add
80
+ result = str(result).strip()
81
+ for prefix in ["The answer is ", "Answer: ", "Final answer: ", "Result: "]:
82
+ if result.lower().startswith(prefix.lower()):
83
+ result = result[len(prefix):]
84
+ return result.strip()
85
+ except Exception as e:
86
+ print(f"Agent error: {e}")
87
+ return f"AGENT ERROR: {e}"
88
+
89
+
90
+ # --- Submission ---
91
+ def run_and_submit_all(profile: gr.OAuthProfile | None):
92
+ space_id = os.getenv("SPACE_ID")
93
 
94
  if profile:
95
+ username = f"{profile.username}"
96
  print(f"User logged in: {username}")
97
  else:
98
  print("User not logged in.")
 
102
  questions_url = f"{api_url}/questions"
103
  submit_url = f"{api_url}/submit"
104
 
 
105
  try:
106
  agent = BasicAgent()
107
  except Exception as e:
108
  print(f"Error instantiating agent: {e}")
109
  return f"Error initializing agent: {e}", None
110
+
111
  agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main"
112
  print(agent_code)
113
 
 
114
  print(f"Fetching questions from: {questions_url}")
115
  try:
116
  response = requests.get(questions_url, timeout=15)
117
  response.raise_for_status()
118
  questions_data = response.json()
119
  if not questions_data:
120
+ print("Fetched questions list is empty.")
121
+ return "Fetched questions list is empty or invalid format.", None
122
  print(f"Fetched {len(questions_data)} questions.")
123
  except requests.exceptions.RequestException as e:
124
  print(f"Error fetching questions: {e}")
125
  return f"Error fetching questions: {e}", None
126
  except requests.exceptions.JSONDecodeError as e:
127
+ print(f"Error decoding JSON response from questions endpoint: {e}")
128
+ return f"Error decoding server response for questions: {e}", None
 
129
  except Exception as e:
130
  print(f"An unexpected error occurred fetching questions: {e}")
131
  return f"An unexpected error occurred fetching questions: {e}", None
132
 
 
133
  results_log = []
134
  answers_payload = []
135
  print(f"Running agent on {len(questions_data)} questions...")
136
+
137
  for item in questions_data:
138
  task_id = item.get("task_id")
139
  question_text = item.get("question")
 
141
  print(f"Skipping item with missing task_id or question: {item}")
142
  continue
143
  try:
144
+ submitted_answer = agent(question_text, task_id)
145
  answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer})
146
  results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer})
147
  except Exception as e:
148
+ print(f"Error running agent on task {task_id}: {e}")
149
+ results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"})
150
 
151
  if not answers_payload:
152
  print("Agent did not produce any answers to submit.")
153
  return "Agent did not produce any answers to submit.", pd.DataFrame(results_log)
154
 
 
155
  submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload}
156
  status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..."
157
  print(status_update)
158
 
 
159
  print(f"Submitting {len(answers_payload)} answers to: {submit_url}")
160
  try:
161
  response = requests.post(submit_url, json=submission_data, timeout=60)
 
199
  return status_message, results_df
200
 
201
 
202
+ # --- Gradio Interface ---
203
  with gr.Blocks() as demo:
204
  gr.Markdown("# Basic Agent Evaluation Runner")
205
  gr.Markdown(
206
  """
207
  **Instructions:**
 
208
  1. Please clone this space, then modify the code to define your agent's logic, the tools, the necessary packages, etc ...
209
  2. Log in to your Hugging Face account using the button below. This uses your HF username for submission.
210
  3. Click 'Run Evaluation & Submit All Answers' to fetch questions, run your agent, submit answers, and see the score.
 
211
  ---
212
  **Disclaimers:**
213
  Once clicking on the "submit button, it can take quite some time ( this is the time for the agent to go through all the questions).
214
+ This space provides a basic setup and is intentionally sub-optimal to encourage you to develop your own, more robust solution.
215
  """
216
  )
217
 
 
220
  run_button = gr.Button("Run Evaluation & Submit All Answers")
221
 
222
  status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False)
 
223
  results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True)
224
 
225
  run_button.click(
 
229
 
230
  if __name__ == "__main__":
231
  print("\n" + "-"*30 + " App Starting " + "-"*30)
 
232
  space_host_startup = os.getenv("SPACE_HOST")
233
+ space_id_startup = os.getenv("SPACE_ID")
234
 
235
  if space_host_startup:
236
+ print(f"SPACE_HOST found: {space_host_startup}")
 
237
  else:
238
+ print("SPACE_HOST not found (running locally?).")
239
 
240
+ if space_id_startup:
241
+ print(f"SPACE_ID found: {space_id_startup}")
242
+ print(f"Repo URL: https://huggingface.co/spaces/{space_id_startup}")
 
243
  else:
244
+ print("SPACE_ID not found (running locally?).")
245
 
246
  print("-"*(60 + len(" App Starting ")) + "\n")
 
247
  print("Launching Gradio Interface for Basic Agent Evaluation...")
248
  demo.launch(debug=True, share=False)