| import os |
| import gradio as gr |
| import requests |
| import inspect |
| import pandas as pd |
| from smolagents import CodeAgent, InferenceClientModel, DuckDuckGoSearchTool,Tool,tool,VisitWebpageTool,PythonInterpreterTool,FinalAnswerTool |
| import base64 |
|
|
| |
| |
| DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space" |
|
|
| |
| |
|
|
| class BasicAgent: |
| def __init__(self): |
| print("Initializing Smolagent...") |
| print("HF_TOKEN present:", bool(os.environ.get("HF_TOKEN"))) |
|
|
| @tool |
| def fetch_task_file(task_id: str) -> str: |
| """ |
| Downloads the file attached to a GAIA task and saves it locally. |
| |
| Args: |
| task_id: The task_id of the current question. |
| |
| Returns: |
| The local file path where the file was saved. |
| """ |
| resp = requests.get(f"{DEFAULT_API_URL}/files/{task_id}") |
| resp.raise_for_status() |
| |
| ext = resp.headers.get("content-type", "").split("/")[-1].split(";")[0] |
| path = f"/tmp/{task_id}.{ext or 'bin'}" |
| with open(path, "wb") as f: |
| f.write(resp.content) |
| return path |
| @tool |
| def get_youtube_transcript(url: str) -> str: |
| """ |
| Fetches the transcript/captions of a YouTube video. |
| |
| Args: |
| url: The full YouTube video URL. |
| |
| Returns: |
| The transcript text, or an error message if unavailable. |
| """ |
| from youtube_transcript_api import YouTubeTranscriptApi |
| import re |
|
|
| match = re.search(r"(?:v=|youtu\.be/)([\w-]{11})", url) |
| if not match: |
| return "Could not extract video ID from URL." |
| video_id = match.group(1) |
|
|
| try: |
| transcript = YouTubeTranscriptApi().fetch(video_id) |
| return " ".join(snippet.text for snippet in transcript) |
| except Exception as e: |
| return f"Transcript unavailable: {e}" |
| |
| @tool |
| def fetch_webpage(url: str) -> str: |
| """ |
| Fetches a webpage and returns its content as markdown. Use this instead of |
| visit_webpage for Wikipedia and other sites that block requests without a |
| real User-Agent header (visit_webpage will get a 403 on many of them). |
| |
| Args: |
| url: The URL to fetch. |
| |
| Returns: |
| The page content converted to markdown, or an error message. |
| """ |
| from markdownify import markdownify |
| import re |
|
|
| headers = { |
| "User-Agent": "GAIA-Agent-Research/1.0 (contact: kaindumushinge@arizona.edu) python-requests" |
| } |
| try: |
| resp = requests.get(url, headers=headers, timeout=20) |
| resp.raise_for_status() |
| content = markdownify(resp.text).strip() |
| content = re.sub(r"\n{3,}", "\n\n", content) |
| return content[:40000] |
| except Exception as e: |
| return f"Error fetching the webpage: {e}" |
|
|
| @tool |
| def read_pdf_from_url(url: str) -> str: |
| """ |
| Downloads a PDF from a URL (e.g. an arXiv paper) and extracts its text. |
| Use this for PDFs reachable by URL; use read_pdf for local files already |
| fetched via fetch_task_file. |
| |
| Args: |
| url: Direct URL to a PDF file. |
| |
| Returns: |
| The extracted text of the PDF, or an error message. |
| """ |
| from pypdf import PdfReader |
| import io |
|
|
| headers = { |
| "User-Agent": "GAIA-Agent-Research/1.0 (contact: kaindumushinge@arizona.edu) python-requests" |
| } |
| try: |
| resp = requests.get(url, headers=headers, timeout=30) |
| resp.raise_for_status() |
| reader = PdfReader(io.BytesIO(resp.content)) |
| return "\n".join(page.extract_text() or "" for page in reader.pages)[:40000] |
| except Exception as e: |
| return f"Error fetching/parsing the PDF: {e}" |
|
|
| @tool |
| def read_spreadsheet(file_path: str) -> str: |
| """ |
| Reads a CSV or Excel file and returns a text summary of its contents. |
| |
| Args: |
| file_path: Local path to the spreadsheet file. |
| |
| Returns: |
| A string representation of the dataframe. |
| """ |
| if file_path.endswith(".csv"): |
| df = pd.read_csv(file_path) |
| else: |
| df = pd.read_excel(file_path) |
| return df.to_string() |
|
|
| class ReadPDFTool(Tool): |
| name = "read_pdf" |
| description = "Extracts text from a PDF file." |
| inputs = { |
| "file_path": { |
| "type": "string", |
| "description": "Local path to the PDF file.", |
| } |
| } |
| output_type = "string" |
|
|
| def forward(self, file_path: str) -> str: |
| from pypdf import PdfReader |
| reader = PdfReader(file_path) |
| return "\n".join(page.extract_text() or "" for page in reader.pages) |
|
|
| class AnalyzeImageTool(Tool): |
| name = "analyze_image" |
| description = "Analyzes an image and answers a question about its contents." |
| inputs = { |
| "file_path": { |
| "type": "string", |
| "description": "Local path to the image file.", |
| }, |
| "question": { |
| "type": "string", |
| "description": "What to look for or answer about the image.", |
| }, |
| } |
| output_type = "string" |
|
|
| def forward(self, file_path: str, question: str) -> str: |
| import base64 |
| from huggingface_hub import InferenceClient |
|
|
| client = InferenceClient(token=os.environ["HF_TOKEN"]) |
| with open(file_path, "rb") as f: |
| image_bytes = f.read() |
| image_b64 = base64.b64encode(image_bytes).decode("utf-8") |
|
|
| result = client.chat_completion( |
| model="Qwen/Qwen2.5-VL-72B-Instruct", |
| messages=[ |
| { |
| "role": "user", |
| "content": [ |
| {"type": "text", "text": question}, |
| {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{image_b64}"}}, |
| ], |
| } |
| ], |
| ) |
| return result.choices[0].message.content |
|
|
|
|
| class TranscribeAudioTool(Tool): |
| name = "transcribe_audio" |
| description = "Transcribes speech from an audio file to text." |
| inputs = { |
| "file_path": { |
| "type": "string", |
| "description": "Local path to the audio file.", |
| } |
| } |
| output_type = "string" |
|
|
| def forward(self, file_path: str) -> str: |
| from huggingface_hub import InferenceClient |
| client = InferenceClient(token=os.environ["HF_TOKEN"]) |
| result = client.automatic_speech_recognition( |
| file_path, |
| model="openai/whisper-large-v3", |
| ) |
| return result.text |
| read_pdf = ReadPDFTool() |
| analyze_image = AnalyzeImageTool() |
| transcribe_audio = TranscribeAudioTool() |
| self.agent = CodeAgent( |
| tools=[ |
| DuckDuckGoSearchTool(), |
| VisitWebpageTool(), |
| fetch_webpage, |
| read_pdf_from_url, |
| PythonInterpreterTool(), |
| FinalAnswerTool(), |
| fetch_task_file, |
| read_pdf, |
| read_spreadsheet, |
| get_youtube_transcript, |
| analyze_image, |
| transcribe_audio, |
| ], |
| model= InferenceClientModel( |
| "Qwen/Qwen3-235B-A22B-Instruct-2507", |
| provider="auto", |
| temperature=0.3, |
| ), |
| additional_authorized_imports=["pandas", "requests", "re", "io"], |
| instructions = ("You are an advanced CodeAgent that will show your capabilities to work in the real world by being tested in GAIA, the agent testing platform. If the question includes a task_id and mentions a file, call fetch_task_file first; route YouTube URLs to get_youtube_transcript, other URLs to fetch_webpage, PDFs at a URL to read_pdf_from_url, local PDFs to read_pdf, and spreadsheets to read_spreadsheet, using web_search only when no URL or file is given,then respond with only the exact final answer value, no explanation, no prefix." |
| "Always call fetch_webpage instead of visit_webpage: visit_webpage sends no User-Agent header and gets " |
| "blocked (403) by Wikipedia and many other sites; fetch_webpage sends a proper header and works reliably. " |
| "Only fall back to visit_webpage if fetch_webpage itself errors." |
| "When a fetched page contains a data table (e.g. Wikipedia infoboxes, Baseball-Reference stat tables), " |
| "prefer pandas.read_html(io.StringIO(page_text)) to extract it as a DataFrame instead of writing regex " |
| "against the raw markdown -- it is far more reliable. If a page's answer is already visible in the text " |
| "you fetched, read it directly rather than writing extraction code." |
| "Use the Thought Action observation to produce high quality results and only answer when you are sure you have performed the necessary steps for the task and question" |
| "If the file is an image, use analyze_image with a specific question about it" |
| "what to find. If the file is audio, use transcribe_audio first, then reason over the transcribed text " |
| "Use the PythonInterpreterTool for code interpretation in python" |
| "NEVER invent, guess, or simulate data you have not actually retrieved. If a " |
| "file cannot be fetched or a page cannot be read, say so explicitly rather " |
| "than fabricating plausible-looking data or answers. " |
| "The grader does an exact string match after light normalization, so format " |
| "the final answer exactly as the question asks: a bare number with no commas, " |
| "units, or currency symbols unless explicitly requested; as few words as " |
| "possible for a string answer, with no articles or explanatory text; and a " |
| "comma-separated list (no surrounding brackets) if multiple items are asked for."), |
| max_steps=20, |
| ) |
| |
| def __call__(self, question: str, task_id: str = None) -> str: |
| print(f"Agent received question: {question[:50]}...") |
| full_prompt = question |
| if task_id: |
| full_prompt = f"{question}\n\n(task_id for this question: {task_id})" |
| try: |
| |
| answer = self.agent.run(full_prompt) |
| return str(answer).strip() |
| except Exception as e: |
| import traceback |
| traceback.print_exc() |
| return "Error" |
|
|
| def run_and_submit_all( profile: gr.OAuthProfile | None): |
| """ |
| Fetches all questions, runs the BasicAgent on them, submits all answers, |
| and displays the results. |
| """ |
| |
| space_id = os.getenv("SPACE_ID") |
|
|
| if profile: |
| username= f"{profile.username}" |
| print(f"User logged in: {username}") |
| else: |
| print("User not logged in.") |
| return "Please Login to Hugging Face with the button.", None |
|
|
| api_url = DEFAULT_API_URL |
| questions_url = f"{api_url}/questions" |
| submit_url = f"{api_url}/submit" |
|
|
| |
| try: |
| agent = BasicAgent() |
| except Exception as e: |
| print(f"Error instantiating agent: {e}") |
| return f"Error initializing agent: {e}", None |
| |
| agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main" |
| print(agent_code) |
|
|
| |
| print(f"Fetching questions from: {questions_url}") |
| try: |
| response = requests.get(questions_url, timeout=15) |
| response.raise_for_status() |
| questions_data = response.json() |
| if not questions_data: |
| print("Fetched questions list is empty.") |
| return "Fetched questions list is empty or invalid format.", None |
| print(f"Fetched {len(questions_data)} questions.") |
| except requests.exceptions.RequestException as e: |
| print(f"Error fetching questions: {e}") |
| return f"Error fetching questions: {e}", None |
| except requests.exceptions.JSONDecodeError as e: |
| print(f"Error decoding JSON response from questions endpoint: {e}") |
| print(f"Response text: {response.text[:500]}") |
| return f"Error decoding server response for questions: {e}", None |
| except Exception as e: |
| print(f"An unexpected error occurred fetching questions: {e}") |
| return f"An unexpected error occurred fetching questions: {e}", None |
|
|
| |
| results_log = [] |
| answers_payload = [] |
| print(f"Running agent on {len(questions_data)} questions...") |
| for item in questions_data: |
| task_id = item.get("task_id") |
| question_text = item.get("question") |
| if not task_id or question_text is None: |
| print(f"Skipping item with missing task_id or question: {item}") |
| continue |
| try: |
| submitted_answer = agent(question_text, task_id) |
| answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer}) |
| results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer}) |
| except Exception as e: |
| print(f"Error running agent on task {task_id}: {e}") |
| results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"}) |
|
|
| if not answers_payload: |
| print("Agent did not produce any answers to submit.") |
| return "Agent did not produce any answers to submit.", pd.DataFrame(results_log) |
|
|
| |
| submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload} |
| status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..." |
| print(status_update) |
|
|
| |
| print(f"Submitting {len(answers_payload)} answers to: {submit_url}") |
| try: |
| response = requests.post(submit_url, json=submission_data, timeout=60) |
| response.raise_for_status() |
| result_data = response.json() |
| final_status = ( |
| f"Submission Successful!\n" |
| f"User: {result_data.get('username')}\n" |
| f"Overall Score: {result_data.get('score', 'N/A')}% " |
| f"({result_data.get('correct_count', '?')}/{result_data.get('total_attempted', '?')} correct)\n" |
| f"Message: {result_data.get('message', 'No message received.')}" |
| ) |
| print("Submission successful.") |
| results_df = pd.DataFrame(results_log) |
| return final_status, results_df |
| except requests.exceptions.HTTPError as e: |
| error_detail = f"Server responded with status {e.response.status_code}." |
| try: |
| error_json = e.response.json() |
| error_detail += f" Detail: {error_json.get('detail', e.response.text)}" |
| except requests.exceptions.JSONDecodeError: |
| error_detail += f" Response: {e.response.text[:500]}" |
| status_message = f"Submission Failed: {error_detail}" |
| print(status_message) |
| results_df = pd.DataFrame(results_log) |
| return status_message, results_df |
| except requests.exceptions.Timeout: |
| status_message = "Submission Failed: The request timed out." |
| print(status_message) |
| results_df = pd.DataFrame(results_log) |
| return status_message, results_df |
| except requests.exceptions.RequestException as e: |
| status_message = f"Submission Failed: Network error - {e}" |
| print(status_message) |
| results_df = pd.DataFrame(results_log) |
| return status_message, results_df |
| except Exception as e: |
| status_message = f"An unexpected error occurred during submission: {e}" |
| print(status_message) |
| results_df = pd.DataFrame(results_log) |
| return status_message, results_df |
|
|
|
|
| |
| with gr.Blocks() as demo: |
| gr.Markdown("# Basic Agent Evaluation Runner") |
| gr.Markdown( |
| """ |
| **Instructions:** |
| |
| 1. Please clone this space, then modify the code to define your agent's logic, the tools, the necessary packages, etc ... |
| 2. Log in to your Hugging Face account using the button below. This uses your HF username for submission. |
| 3. Click 'Run Evaluation & Submit All Answers' to fetch questions, run your agent, submit answers, and see the score. |
| |
| --- |
| **Disclaimers:** |
| Once clicking on the "submit button, it can take quite some time ( this is the time for the agent to go through all the questions). |
| This space provides a basic setup and is intentionally sub-optimal to encourage you to develop your own, more robust solution. For instance for the delay process of the submit button, a solution could be to cache the answers and submit in a seperate action or even to answer the questions in async. |
| """ |
| ) |
|
|
| gr.LoginButton() |
|
|
| run_button = gr.Button("Run Evaluation & Submit All Answers") |
|
|
| status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False) |
| |
| results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True) |
|
|
| run_button.click( |
| fn=run_and_submit_all, |
| outputs=[status_output, results_table] |
| ) |
|
|
| if __name__ == "__main__": |
| print("\n" + "-"*30 + " App Starting " + "-"*30) |
| |
| space_host_startup = os.getenv("SPACE_HOST") |
| space_id_startup = os.getenv("SPACE_ID") |
|
|
| if space_host_startup: |
| print(f"✅ SPACE_HOST found: {space_host_startup}") |
| print(f" Runtime URL should be: https://{space_host_startup}.hf.space") |
| else: |
| print("ℹ️ SPACE_HOST environment variable not found (running locally?).") |
|
|
| if space_id_startup: |
| print(f"✅ SPACE_ID found: {space_id_startup}") |
| print(f" Repo URL: https://huggingface.co/spaces/{space_id_startup}") |
| print(f" Repo Tree URL: https://huggingface.co/spaces/{space_id_startup}/tree/main") |
| else: |
| print("ℹ️ SPACE_ID environment variable not found (running locally?). Repo URL cannot be determined.") |
|
|
| print("-"*(60 + len(" App Starting ")) + "\n") |
|
|
| print("Launching Gradio Interface for Basic Agent Evaluation...") |
| demo.launch(debug=True, share=False) |