tobyvertommen Claude Sonnet 4.6 commited on
Commit
fbc830e
·
1 Parent(s): 81917a3

Add LangGraph + Groq ReAct agent for GAIA benchmark

Browse files

Replace placeholder BasicAgent with LangGraph ReAct agent using Groq
(llama-3.1-70b-versatile). Tools: DuckDuckGo, Wikipedia, Python REPL,
GAIA file fetcher. System prompt enforces exact-match answer format.

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>

Files changed (2) hide show
  1. app.py +94 -58
  2. requirements.txt +9 -1
app.py CHANGED
@@ -1,34 +1,81 @@
1
  import os
2
  import gradio as gr
3
  import requests
4
- import inspect
5
  import pandas as pd
 
 
 
 
 
 
6
 
7
- # (Keep Constants as is)
8
  # --- Constants ---
9
  DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
10
 
11
- # --- Basic Agent Definition ---
12
- # ----- THIS IS WERE YOU CAN BUILD WHAT YOU WANT ------
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
13
  class BasicAgent:
14
  def __init__(self):
15
- print("BasicAgent initialized.")
16
- def __call__(self, question: str) -> str:
17
- print(f"Agent received question (first 50 chars): {question[:50]}...")
18
- fixed_answer = "This is a default answer."
19
- print(f"Agent returning fixed answer: {fixed_answer}")
20
- return fixed_answer
21
-
22
- def run_and_submit_all( profile: gr.OAuthProfile | None):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
23
  """
24
  Fetches all questions, runs the BasicAgent on them, submits all answers,
25
  and displays the results.
26
  """
27
- # --- Determine HF Space Runtime URL and Repo URL ---
28
- space_id = os.getenv("SPACE_ID") # Get the SPACE_ID for sending link to the code
29
 
30
  if profile:
31
- username= f"{profile.username}"
32
  print(f"User logged in: {username}")
33
  else:
34
  print("User not logged in.")
@@ -38,13 +85,13 @@ def run_and_submit_all( profile: gr.OAuthProfile | None):
38
  questions_url = f"{api_url}/questions"
39
  submit_url = f"{api_url}/submit"
40
 
41
- # 1. Instantiate Agent ( modify this part to create your agent)
42
  try:
43
  agent = BasicAgent()
44
  except Exception as e:
45
  print(f"Error instantiating agent: {e}")
46
  return f"Error initializing agent: {e}", None
47
- # In the case of an app running as a hugging Face space, this link points toward your codebase ( usefull for others so please keep it public)
48
  agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main"
49
  print(agent_code)
50
 
@@ -55,21 +102,21 @@ def run_and_submit_all( profile: gr.OAuthProfile | None):
55
  response.raise_for_status()
56
  questions_data = response.json()
57
  if not questions_data:
58
- print("Fetched questions list is empty.")
59
- return "Fetched questions list is empty or invalid format.", None
60
  print(f"Fetched {len(questions_data)} questions.")
61
  except requests.exceptions.RequestException as e:
62
  print(f"Error fetching questions: {e}")
63
  return f"Error fetching questions: {e}", None
64
  except requests.exceptions.JSONDecodeError as e:
65
- print(f"Error decoding JSON response from questions endpoint: {e}")
66
- print(f"Response text: {response.text[:500]}")
67
- return f"Error decoding server response for questions: {e}", None
68
  except Exception as e:
69
  print(f"An unexpected error occurred fetching questions: {e}")
70
  return f"An unexpected error occurred fetching questions: {e}", None
71
 
72
- # 3. Run your Agent
73
  results_log = []
74
  answers_payload = []
75
  print(f"Running agent on {len(questions_data)} questions...")
@@ -80,23 +127,22 @@ def run_and_submit_all( profile: gr.OAuthProfile | None):
80
  print(f"Skipping item with missing task_id or question: {item}")
81
  continue
82
  try:
83
- submitted_answer = agent(question_text)
84
  answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer})
85
  results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer})
86
  except Exception as e:
87
- print(f"Error running agent on task {task_id}: {e}")
88
- results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"})
89
 
90
  if not answers_payload:
91
  print("Agent did not produce any answers to submit.")
92
  return "Agent did not produce any answers to submit.", pd.DataFrame(results_log)
93
 
94
- # 4. Prepare Submission
95
  submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload}
96
  status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..."
97
  print(status_update)
98
 
99
- # 5. Submit
100
  print(f"Submitting {len(answers_payload)} answers to: {submit_url}")
101
  try:
102
  response = requests.post(submit_url, json=submission_data, timeout=60)
@@ -110,8 +156,7 @@ def run_and_submit_all( profile: gr.OAuthProfile | None):
110
  f"Message: {result_data.get('message', 'No message received.')}"
111
  )
112
  print("Submission successful.")
113
- results_df = pd.DataFrame(results_log)
114
- return final_status, results_df
115
  except requests.exceptions.HTTPError as e:
116
  error_detail = f"Server responded with status {e.response.status_code}."
117
  try:
@@ -121,40 +166,36 @@ def run_and_submit_all( profile: gr.OAuthProfile | None):
121
  error_detail += f" Response: {e.response.text[:500]}"
122
  status_message = f"Submission Failed: {error_detail}"
123
  print(status_message)
124
- results_df = pd.DataFrame(results_log)
125
- return status_message, results_df
126
  except requests.exceptions.Timeout:
127
  status_message = "Submission Failed: The request timed out."
128
  print(status_message)
129
- results_df = pd.DataFrame(results_log)
130
- return status_message, results_df
131
  except requests.exceptions.RequestException as e:
132
  status_message = f"Submission Failed: Network error - {e}"
133
  print(status_message)
134
- results_df = pd.DataFrame(results_log)
135
- return status_message, results_df
136
  except Exception as e:
137
  status_message = f"An unexpected error occurred during submission: {e}"
138
  print(status_message)
139
- results_df = pd.DataFrame(results_log)
140
- return status_message, results_df
141
 
142
 
143
- # --- Build Gradio Interface using Blocks ---
144
  with gr.Blocks() as demo:
145
- gr.Markdown("# Basic Agent Evaluation Runner")
146
  gr.Markdown(
147
  """
148
  **Instructions:**
149
 
150
- 1. Please clone this space, then modify the code to define your agent's logic, the tools, the necessary packages, etc ...
151
- 2. Log in to your Hugging Face account using the button below. This uses your HF username for submission.
152
- 3. Click 'Run Evaluation & Submit All Answers' to fetch questions, run your agent, submit answers, and see the score.
 
 
153
 
154
  ---
155
- **Disclaimers:**
156
- Once clicking on the "submit button, it can take quite some time ( this is the time for the agent to go through all the questions).
157
- This space provides a basic setup and is intentionally sub-optimal to encourage you to develop your own, more robust solution. For instance for the delay process of the submit button, a solution could be to cache the answers and submit in a seperate action or even to answer the questions in async.
158
  """
159
  )
160
 
@@ -163,7 +204,6 @@ with gr.Blocks() as demo:
163
  run_button = gr.Button("Run Evaluation & Submit All Answers")
164
 
165
  status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False)
166
- # Removed max_rows=10 from DataFrame constructor
167
  results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True)
168
 
169
  run_button.click(
@@ -172,25 +212,21 @@ with gr.Blocks() as demo:
172
  )
173
 
174
  if __name__ == "__main__":
175
- print("\n" + "-"*30 + " App Starting " + "-"*30)
176
- # Check for SPACE_HOST and SPACE_ID at startup for information
177
  space_host_startup = os.getenv("SPACE_HOST")
178
- space_id_startup = os.getenv("SPACE_ID") # Get SPACE_ID at startup
179
 
180
  if space_host_startup:
181
  print(f"✅ SPACE_HOST found: {space_host_startup}")
182
  print(f" Runtime URL should be: https://{space_host_startup}.hf.space")
183
  else:
184
- print("ℹ️ SPACE_HOST environment variable not found (running locally?).")
185
 
186
- if space_id_startup: # Print repo URLs if SPACE_ID is found
187
  print(f"✅ SPACE_ID found: {space_id_startup}")
188
  print(f" Repo URL: https://huggingface.co/spaces/{space_id_startup}")
189
- print(f" Repo Tree URL: https://huggingface.co/spaces/{space_id_startup}/tree/main")
190
  else:
191
- print("ℹ️ SPACE_ID environment variable not found (running locally?). Repo URL cannot be determined.")
192
-
193
- print("-"*(60 + len(" App Starting ")) + "\n")
194
 
195
- print("Launching Gradio Interface for Basic Agent Evaluation...")
196
- demo.launch(debug=True, share=False)
 
1
  import os
2
  import gradio as gr
3
  import requests
 
4
  import pandas as pd
5
+ from langchain_groq import ChatGroq
6
+ from langgraph.prebuilt import create_react_agent
7
+ from langchain_community.tools import DuckDuckGoSearchRun, WikipediaQueryRun
8
+ from langchain_community.utilities import WikipediaAPIWrapper
9
+ from langchain_experimental.tools import PythonREPLTool
10
+ from langchain.tools import tool
11
 
 
12
  # --- Constants ---
13
  DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
14
 
15
+ SYSTEM_PROMPT = """You are a general AI assistant tasked with answering questions accurately.
16
+
17
+ Reason step by step, then finish your answer with exactly this format:
18
+ FINAL ANSWER: [your answer]
19
+
20
+ Rules for FINAL ANSWER:
21
+ - Numbers: no commas, no units unless specified
22
+ - Strings: no articles (a/an/the), no abbreviations, write digits in plain text
23
+ - Lists: comma-separated, apply above rules per element
24
+ - Be precise and concise — the answer is evaluated by exact match"""
25
+
26
+
27
+ @tool
28
+ def fetch_task_file(task_id: str) -> str:
29
+ """Fetch a file associated with a GAIA task by its task_id. Use this when the question references an attachment, file, or additional data."""
30
+ try:
31
+ response = requests.get(f"{DEFAULT_API_URL}/files/{task_id}", timeout=15)
32
+ if response.status_code == 200:
33
+ content_type = response.headers.get("content-type", "")
34
+ if any(t in content_type for t in ["text", "json", "csv", "xml"]):
35
+ return response.text[:8000]
36
+ return f"Binary file ({content_type}) — cannot read as text"
37
+ return "No file found for this task"
38
+ except Exception as e:
39
+ return f"Error fetching file: {e}"
40
+
41
+
42
  class BasicAgent:
43
  def __init__(self):
44
+ llm = ChatGroq(model="llama-3.1-70b-versatile", temperature=0)
45
+ tools = [
46
+ DuckDuckGoSearchRun(),
47
+ WikipediaQueryRun(api_wrapper=WikipediaAPIWrapper(top_k_results=3)),
48
+ PythonREPLTool(),
49
+ fetch_task_file,
50
+ ]
51
+ self.agent = create_react_agent(llm, tools, state_modifier=SYSTEM_PROMPT)
52
+ print("BasicAgent initialized with LangGraph + Groq (llama-3.1-70b-versatile).")
53
+
54
+ def __call__(self, question: str, task_id: str = "") -> str:
55
+ full_question = f"[Task ID: {task_id}]\n\n{question}" if task_id else question
56
+ print(f"Running agent on task {task_id}: {question[:80]}...")
57
+ result = self.agent.invoke(
58
+ {"messages": [("user", full_question)]},
59
+ {"recursion_limit": 50},
60
+ )
61
+ raw_answer = result["messages"][-1].content
62
+ if "FINAL ANSWER:" in raw_answer:
63
+ answer = raw_answer.split("FINAL ANSWER:")[-1].strip()
64
+ else:
65
+ answer = raw_answer.strip()
66
+ print(f"Answer for {task_id}: {answer}")
67
+ return answer
68
+
69
+
70
+ def run_and_submit_all(profile: gr.OAuthProfile | None):
71
  """
72
  Fetches all questions, runs the BasicAgent on them, submits all answers,
73
  and displays the results.
74
  """
75
+ space_id = os.getenv("SPACE_ID")
 
76
 
77
  if profile:
78
+ username = f"{profile.username}"
79
  print(f"User logged in: {username}")
80
  else:
81
  print("User not logged in.")
 
85
  questions_url = f"{api_url}/questions"
86
  submit_url = f"{api_url}/submit"
87
 
88
+ # 1. Instantiate Agent
89
  try:
90
  agent = BasicAgent()
91
  except Exception as e:
92
  print(f"Error instantiating agent: {e}")
93
  return f"Error initializing agent: {e}", None
94
+
95
  agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main"
96
  print(agent_code)
97
 
 
102
  response.raise_for_status()
103
  questions_data = response.json()
104
  if not questions_data:
105
+ print("Fetched questions list is empty.")
106
+ return "Fetched questions list is empty or invalid format.", None
107
  print(f"Fetched {len(questions_data)} questions.")
108
  except requests.exceptions.RequestException as e:
109
  print(f"Error fetching questions: {e}")
110
  return f"Error fetching questions: {e}", None
111
  except requests.exceptions.JSONDecodeError as e:
112
+ print(f"Error decoding JSON response from questions endpoint: {e}")
113
+ print(f"Response text: {response.text[:500]}")
114
+ return f"Error decoding server response for questions: {e}", None
115
  except Exception as e:
116
  print(f"An unexpected error occurred fetching questions: {e}")
117
  return f"An unexpected error occurred fetching questions: {e}", None
118
 
119
+ # 3. Run Agent
120
  results_log = []
121
  answers_payload = []
122
  print(f"Running agent on {len(questions_data)} questions...")
 
127
  print(f"Skipping item with missing task_id or question: {item}")
128
  continue
129
  try:
130
+ submitted_answer = agent(question_text, task_id)
131
  answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer})
132
  results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer})
133
  except Exception as e:
134
+ print(f"Error running agent on task {task_id}: {e}")
135
+ results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"})
136
 
137
  if not answers_payload:
138
  print("Agent did not produce any answers to submit.")
139
  return "Agent did not produce any answers to submit.", pd.DataFrame(results_log)
140
 
141
+ # 4. Submit
142
  submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload}
143
  status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..."
144
  print(status_update)
145
 
 
146
  print(f"Submitting {len(answers_payload)} answers to: {submit_url}")
147
  try:
148
  response = requests.post(submit_url, json=submission_data, timeout=60)
 
156
  f"Message: {result_data.get('message', 'No message received.')}"
157
  )
158
  print("Submission successful.")
159
+ return final_status, pd.DataFrame(results_log)
 
160
  except requests.exceptions.HTTPError as e:
161
  error_detail = f"Server responded with status {e.response.status_code}."
162
  try:
 
166
  error_detail += f" Response: {e.response.text[:500]}"
167
  status_message = f"Submission Failed: {error_detail}"
168
  print(status_message)
169
+ return status_message, pd.DataFrame(results_log)
 
170
  except requests.exceptions.Timeout:
171
  status_message = "Submission Failed: The request timed out."
172
  print(status_message)
173
+ return status_message, pd.DataFrame(results_log)
 
174
  except requests.exceptions.RequestException as e:
175
  status_message = f"Submission Failed: Network error - {e}"
176
  print(status_message)
177
+ return status_message, pd.DataFrame(results_log)
 
178
  except Exception as e:
179
  status_message = f"An unexpected error occurred during submission: {e}"
180
  print(status_message)
181
+ return status_message, pd.DataFrame(results_log)
 
182
 
183
 
184
+ # --- Gradio Interface ---
185
  with gr.Blocks() as demo:
186
+ gr.Markdown("# Agent Evaluation Runner — LangGraph + Groq")
187
  gr.Markdown(
188
  """
189
  **Instructions:**
190
 
191
+ 1. Log in to your Hugging Face account using the button below.
192
+ 2. Click 'Run Evaluation & Submit All Answers' to fetch questions, run the agent, and submit.
193
+
194
+ **Agent:** LangGraph ReAct agent with Groq (llama-3.1-70b-versatile)
195
+ **Tools:** DuckDuckGo search, Wikipedia, Python REPL, File fetcher
196
 
197
  ---
198
+ *Note: Running 20 questions takes several minutes.*
 
 
199
  """
200
  )
201
 
 
204
  run_button = gr.Button("Run Evaluation & Submit All Answers")
205
 
206
  status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False)
 
207
  results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True)
208
 
209
  run_button.click(
 
212
  )
213
 
214
  if __name__ == "__main__":
215
+ print("\n" + "-" * 30 + " App Starting " + "-" * 30)
 
216
  space_host_startup = os.getenv("SPACE_HOST")
217
+ space_id_startup = os.getenv("SPACE_ID")
218
 
219
  if space_host_startup:
220
  print(f"✅ SPACE_HOST found: {space_host_startup}")
221
  print(f" Runtime URL should be: https://{space_host_startup}.hf.space")
222
  else:
223
+ print("ℹ️ SPACE_HOST not found (running locally?).")
224
 
225
+ if space_id_startup:
226
  print(f"✅ SPACE_ID found: {space_id_startup}")
227
  print(f" Repo URL: https://huggingface.co/spaces/{space_id_startup}")
 
228
  else:
229
+ print("ℹ️ SPACE_ID not found (running locally?).")
 
 
230
 
231
+ print("-" * (60 + len(" App Starting ")) + "\n")
232
+ demo.launch(debug=True, share=False)
requirements.txt CHANGED
@@ -1,2 +1,10 @@
1
  gradio
2
- requests
 
 
 
 
 
 
 
 
 
1
  gradio
2
+ requests
3
+ pandas
4
+ langchain-groq
5
+ langgraph
6
+ langchain-community
7
+ langchain-experimental
8
+ langchain
9
+ duckduckgo-search
10
+ wikipedia