import os import time import gradio as gr import requests import pandas as pd from smolagents import CodeAgent, OpenAIServerModel, PythonInterpreterTool, Tool from smolagents import FinalAnswerTool # --- Constants --- DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space" # --- Custom Tools --- class WebSearchTool(Tool): name = "web_search" description = "Search the web for information. Use for any factual question." inputs = {"query": {"type": "string", "description": "The search query"}} output_type = "string" def forward(self, query: str) -> str: try: from ddgs import DDGS with DDGS() as ddgs: results = list(ddgs.text(query, max_results=5)) if not results: return "No results found." output = "" for r in results: output += f"Title: {r.get('title', '')}\n" output += f"URL: {r.get('href', '')}\n" output += f"Summary: {r.get('body', '')}\n\n" return output[:3000] except Exception as e: return f"Search error: {e}" class WikipediaTool(Tool): name = "wikipedia_search" description = "Search Wikipedia directly. Use when the question mentions Wikipedia or needs encyclopedic facts like discographies, biographies, lists." inputs = {"query": {"type": "string", "description": "The Wikipedia article title or topic to search"}} output_type = "string" def forward(self, query: str) -> str: try: # First search for the right article search_url = ( "https://en.wikipedia.org/w/api.php" f"?action=query&list=search&srsearch={requests.utils.quote(query)}" "&format=json&srlimit=1" ) r = requests.get(search_url, timeout=10) results = r.json()["query"]["search"] if not results: return "No Wikipedia article found." title = results[0]["title"] # Then fetch full article text content_url = ( "https://en.wikipedia.org/w/api.php" f"?action=query&titles={requests.utils.quote(title)}" "&prop=extracts&explaintext=true&format=json" ) r2 = requests.get(content_url, timeout=10) pages = r2.json()["query"]["pages"] page = next(iter(pages.values())) text = page.get("extract", "No content found") return f"Article: {title}\n\n{text[:5000]}" except Exception as e: return f"Wikipedia error: {e}" class YouTubeTranscriptTool(Tool): name = "youtube_transcript" description = "Gets the transcript/captions of a YouTube video. Use when the question contains a YouTube URL." inputs = {"url": {"type": "string", "description": "YouTube video URL or video ID"}} output_type = "string" def forward(self, url: str) -> str: try: from youtube_transcript_api import YouTubeTranscriptApi if "v=" in url: video_id = url.split("v=")[1].split("&")[0] elif "youtu.be/" in url: video_id = url.split("youtu.be/")[1].split("?")[0] else: video_id = url.strip() ytt = YouTubeTranscriptApi() transcript = ytt.fetch(video_id) return " ".join([t.text for t in transcript])[:5000] except Exception as e: return f"Transcript error: {e}" class FileDownloadTool(Tool): name = "download_file" description = "Downloads a file attached to a GAIA question using its task_id. Use when the question references an attached file, image, CSV, or PDF." inputs = {"task_id": {"type": "string", "description": "The task_id of the current question"}} output_type = "string" def forward(self, task_id: str) -> str: try: url = f"https://agents-course-unit4-scoring.hf.space/files/{task_id}" r = requests.get(url, timeout=15) if r.status_code == 200: return r.text[:5000] return f"No file found for task_id {task_id}" except Exception as e: return f"File download error: {e}" class VisitWebpageTool(Tool): name = "visit_webpage" description = "Fetches the full content of a webpage given its URL. Use when you have a specific URL to read." inputs = {"url": {"type": "string", "description": "The URL of the webpage to visit"}} output_type = "string" def forward(self, url: str) -> str: try: headers = {"User-Agent": "Mozilla/5.0"} r = requests.get(url, timeout=10, headers=headers) # strip html tags roughly import re text = re.sub(r'<[^>]+>', ' ', r.text) text = re.sub(r'\s+', ' ', text).strip() return text[:5000] except Exception as e: return f"Webpage error: {e}" # --- Agent --- class BasicAgent: def __init__(self): model = OpenAIServerModel( model_id="meta-llama/llama-4-scout-17b-16e-instruct", api_base="https://api.groq.com/openai/v1", api_key=os.getenv("GROQ_API_KEY") ) self.agent = CodeAgent( # <-- back to CodeAgent model=model, tools=[ WebSearchTool(), WikipediaTool(), YouTubeTranscriptTool(), FileDownloadTool(), VisitWebpageTool(), PythonInterpreterTool(), ], max_steps=6, ) def __call__(self, question: str, task_id: str = "") -> str: try: prompt = f"""Answer the following question accurately. Return ONLY the final answer with no explanation, no punctuation, no extra words. - If the answer is a number, return just the number. - If the answer is a name, return just the name. - If the answer is a list, return comma separated values in alphabetical order. - If the question asks about a YouTube video, use the youtube_transcript tool. - If the question mentions Wikipedia, use the wikipedia_search tool. - If the question references an attached file, use download_file with the task_id below. Task ID: {task_id} Question: {question}""" result = self.agent.run(prompt) if isinstance(result, list): for block in result: if isinstance(block, dict) and block.get('type') == 'text': return block['text'].strip() return str(result).strip() except Exception as e: print(f"Agent error: {e}") return "I don't know" # --- Main Evaluation Function --- def run_and_submit_all(profile: gr.OAuthProfile | None): space_id = os.getenv("SPACE_ID") if profile: username = f"{profile.username}" print(f"User logged in: {username}") else: return "Please Login to Hugging Face with the button.", None api_url = DEFAULT_API_URL questions_url = f"{api_url}/questions" submit_url = f"{api_url}/submit" try: agent = BasicAgent() except Exception as e: return f"Error initializing agent: {e}", None agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main" print(agent_code) print(f"Fetching questions from: {questions_url}") try: response = requests.get(questions_url, timeout=15) response.raise_for_status() questions_data = response.json() if not questions_data: return "Fetched questions list is empty or invalid format.", None print(f"Fetched {len(questions_data)} questions.") except Exception as e: return f"Error fetching questions: {e}", None results_log = [] answers_payload = [] print(f"Running agent on {len(questions_data)} questions...") for i, item in enumerate(questions_data): task_id = item.get("task_id") question_text = item.get("question") if not task_id or question_text is None: print(f"Skipping item with missing task_id or question: {item}") continue print(f"\n[{i+1}/{len(questions_data)}] Task: {task_id}") print(f"Question: {question_text[:120]}...") try: submitted_answer = agent(question_text, task_id) print(f"Answer: {submitted_answer}") answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer}) results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer}) except Exception as e: print(f"Error on task {task_id}: {e}") results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"}) if i < len(questions_data) - 1: print("Waiting 15s for rate limits...") time.sleep(15) if not answers_payload: return "Agent did not produce any answers to submit.", pd.DataFrame(results_log) submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload} print(f"\nSubmitting {len(answers_payload)} answers...") try: response = requests.post(submit_url, json=submission_data, timeout=60) response.raise_for_status() result_data = response.json() final_status = ( f"Submission Successful!\n" f"User: {result_data.get('username')}\n" f"Overall Score: {result_data.get('score', 'N/A')}% " f"({result_data.get('correct_count', '?')}/{result_data.get('total_attempted', '?')} correct)\n" f"Message: {result_data.get('message', 'No message received.')}" ) print("Submission successful.") return final_status, pd.DataFrame(results_log) except requests.exceptions.HTTPError as e: error_detail = f"Server responded with status {e.response.status_code}." try: error_json = e.response.json() error_detail += f" Detail: {error_json.get('detail', e.response.text)}" except Exception: error_detail += f" Response: {e.response.text[:500]}" return f"Submission Failed: {error_detail}", pd.DataFrame(results_log) except Exception as e: return f"Submission error: {e}", pd.DataFrame(results_log) # --- Gradio UI --- with gr.Blocks() as demo: gr.Markdown("# GAIA Agent Evaluation Runner") gr.Markdown( """ **Instructions:** 1. Log in to your Hugging Face account using the button below. 2. Click 'Run Evaluation & Submit All Answers' to start. 3. Takes ~6 minutes for all 20 questions due to rate limits. """ ) gr.LoginButton() run_button = gr.Button("Run Evaluation & Submit All Answers") status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False) results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True) run_button.click( fn=run_and_submit_all, outputs=[status_output, results_table] ) if __name__ == "__main__": print("\n" + "-"*30 + " App Starting " + "-"*30) space_host_startup = os.getenv("SPACE_HOST") space_id_startup = os.getenv("SPACE_ID") if space_host_startup: print(f"✅ SPACE_HOST found: {space_host_startup}") else: print("ℹ️ SPACE_HOST not found (running locally).") if space_id_startup: print(f"✅ SPACE_ID found: {space_id_startup}") else: print("ℹ️ SPACE_ID not found (running locally).") print("-"*(60 + len(" App Starting ")) + "\n") print("Launching Gradio Interface...") demo.launch(debug=True, share=False)