| import os |
| import gradio as gr |
| import requests |
| import inspect |
| import pandas as pd |
| import re |
| from PyPDF2 import PdfReader |
| from io import BytesIO |
| import yaml |
| from smolagents import CodeAgent, InferenceClientModel, DuckDuckGoSearchTool, tool, PromptTemplates, VisitWebpageTool |
| import wikipedia |
| from bs4 import BeautifulSoup |
| import pdfplumber |
| from youtube_transcript_api import YouTubeTranscriptApi |
| from sportsreference.mlb.roster import Player |
| import whisper |
| import sys, io |
|
|
|
|
|
|
| |
| |
| DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space" |
|
|
| |
|
|
| import wikipedia |
|
|
|
|
| @tool |
| def exec_python(code: str) -> str: |
| """ |
| Executes a short Python code snippet in a sandboxed environment and returns the standard output or any runtime error. |
| |
| Args: |
| code (str): A string containing valid Python code. |
| Typically limited to a few lines of computation, function calls, or print statements. |
| Example: 'for i in range(3): print(i ** 2)' |
| |
| Returns: |
| str: The result printed by the code, or an error message if execution fails. |
| If the code runs but produces no output, returns 'None'. |
| |
| Behavior: |
| - Captures all output from `print()` statements. |
| - Overrides the built-in `print` function to redirect output to a buffer. |
| - Does not persist variables between calls (runs in isolated scope). |
| - No access to file I/O, network, or external libraries unless included in code string. |
| |
| Security Notes: |
| - This function does not use `eval()` and limits execution scope to an empty global environment. |
| - However, because it still uses `exec()`, it should not be exposed to untrusted users without additional sandboxing. |
| |
| Raises: |
| None directly, but returns a string with an error message if execution throws an exception. |
| |
| Example: |
| >>> exec_python('for i in range(3): print(i * 2)') |
| '0\n2\n4' |
| |
| >>> exec_python('x = 5 / 0') |
| 'Error: division by zero' |
| """ |
| buffer = io.StringIO() |
| try: |
| exec(code, {}, {"print": lambda *args: buffer.write(" ".join(str(a) for a in args) + "\n")}) |
| return buffer.getvalue().strip() or "None" |
| except Exception as e: |
| return f"Error: {e}" |
|
|
|
|
|
|
| @tool |
| def youtube_transcript(url: str) -> str: |
| """ |
| Fetches and concatenates the subtitles/transcript of a YouTube video. |
| Args: |
| url (str): Full YouTube video URL. |
| Returns: |
| str: Plain-text transcript of the entire video. |
| Raises: |
| ValueError: If no transcript is available or video ID is invalid. |
| """ |
| vid = url.split("v=")[-1] |
| entries = YouTubeTranscriptApi.get_transcript(vid) |
| return " ".join(item["text"] for item in entries) |
|
|
| @tool |
| def wiki_lookup(question: str) -> str: |
| """ |
| Looks up relevant content from Wikipedia based on the question. |
| Args: |
| question (str): What to look up. |
| Returns: |
| str: Summary content from the most relevant Wikipedia page. |
| """ |
| try: |
| wikipedia.set_lang("en") |
| results = wikipedia.search(question, results=1) |
| if not results: |
| return "❌ No relevant Wikipedia page found." |
| page = wikipedia.page(results[0]) |
| return page.content[:1000] |
| except Exception as e: |
| return f"❌ Wikipedia lookup failed: {e}" |
|
|
| @tool |
| def transcribe_audio(data: bytes) -> str: |
| """ |
| Transcribes raw audio bytes using Whisper. |
| Args: |
| data (bytes): MP3 or WAV file content. |
| Returns: |
| str: Full transcription. |
| """ |
| model = whisper.load_model("base") |
| result = model.transcribe(data) |
| return result["text"].strip() |
|
|
|
|
| @tool |
| def baseball_stats(player_query: str, season: int, stat: str) -> dict: |
| """ |
| Fetches baseball stats from SportsReference.com. |
| Args: |
| player_query (str): Player name or filter. |
| season (int): Year of season. |
| stat (str): Which stat (“walks”, “at_bats”, etc.). |
| Returns: |
| Dict with requested numbers. |
| """ |
| p = Player(player_query, year=season) |
| return {stat: getattr(p, stat)} |
|
|
|
|
| @tool |
| def text_reverse(text: str) -> str: |
| """ |
| Reverses the input string character-by-character. |
| |
| Args: |
| text (str): The input text to reverse. |
| |
| Returns: |
| str: The reversed text. |
| """ |
| return text[::-1] |
|
|
| @tool |
| def pdf_scraper(pdf_url: str) -> str: |
| """ |
| Downloads and extracts text from a PDF at a given URL. |
| |
| Args: |
| pdf_url (str): Direct link to the PDF. |
| |
| Returns: |
| str: Full extracted text from the PDF. |
| """ |
| response = requests.get(pdf_url) |
| reader = PdfReader(BytesIO(response.content)) |
| text = "\n".join(page.extract_text() or '' for page in reader.pages) |
| return text |
|
|
| @tool |
| def parse_excel(data: bytes, sheet_name: str = None) -> str: |
| """ |
| Parses an Excel file from raw bytes and converts it to a CSV-formatted string. |
| |
| This tool reads the specified sheet or the first sheet by default, |
| preserves column headers, and omits row indices in the output. |
| |
| Args: |
| data (bytes): Raw bytes content of the uploaded Excel file (.xlsx, .xls). |
| sheet_name (str, optional): Name or index of the sheet to read. |
| If None, defaults to the first sheet in the workbook. |
| |
| Returns: |
| str: CSV-formatted text of the sheet's data with headers and without indices. |
| |
| Raises: |
| ValueError: If the provided data is not a valid Excel file. |
| XLRDError: If the specified sheet_name does not exist. |
| """ |
| try: |
| df = pd.read_excel(BytesIO(data), sheet_name=sheet_name) |
| except Exception as e: |
| raise ValueError(f"Failed to parse Excel data: {e}") |
| return df.to_csv(index=False) |
|
|
|
|
|
|
| |
| class BasicAgent: |
| def __init__(self): |
| print("🔧 SmolAgent is being initialized.") |
| |
| model = InferenceClientModel( |
| model_id='meta-llama/Llama-3.3-70B-Instruct') |
| |
| self.agent = CodeAgent( |
| model=model, |
| tools=[text_reverse, pdf_scraper, |
| DuckDuckGoSearchTool(), wiki_lookup, VisitWebpageTool(),parse_excel, baseball_stats, youtube_transcript, transcribe_audio, exec_python], |
| add_base_tools=True, |
| planning_interval=2 |
| ) |
|
|
| def __call__(self, question: str) -> str: |
| print(f"🤖 Agent received question: {question[:80]}...") |
| try: |
| result = self.agent.run(question) |
| print(f"✅ Agent result: {result}") |
| return result |
| except Exception as e: |
| error_message = f"❌ Error during agent execution: {e}" |
| print(error_message) |
| return error_message |
|
|
|
|
| |
| def test_random_question(): |
| try: |
| |
| res = requests.get(f"{DEFAULT_API_URL}/random-question") |
| if res.status_code != 200: |
| return "❌ Failed to fetch random question.", "", "" |
| q = res.json() |
| question = q["question"] |
| task_id = q["task_id"] |
| print(f"🎯 Testing on task_id: {task_id}") |
|
|
| |
| agent = BasicAgent() |
| answer = agent(question) |
|
|
| |
| payload = {"task_id": task_id, "answer": answer} |
| eval_res = requests.post(f"{DEFAULT_API_URL}/evaluate", json=payload) |
| if eval_res.status_code != 200: |
| return question, answer, "❌ Evaluation failed." |
|
|
| score = eval_res.json().get("score", "No score returned") |
| return question, answer, f"✅ Score: {score}" |
| |
| except Exception as e: |
| return "❌ Error occurred.", "", f"⚠️ {str(e)}" |
| |
| def run_and_submit_all( profile: gr.OAuthProfile | None): |
| """ |
| Fetches all questions, runs the BasicAgent on them, submits all answers, |
| and displays the results. |
| """ |
| |
| space_id = os.getenv("SPACE_ID") |
|
|
| if profile: |
| username= f"{profile.username}" |
| print(f"User logged in: {username}") |
| else: |
| print("User not logged in.") |
| return "Please Login to Hugging Face with the button.", None |
|
|
| api_url = DEFAULT_API_URL |
| questions_url = f"{api_url}/questions" |
| submit_url = f"{api_url}/submit" |
|
|
| |
| try: |
| agent = BasicAgent() |
| except Exception as e: |
| print(f"Error instantiating agent: {e}") |
| return f"Error initializing agent: {e}", None |
| |
| agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main" |
| print(agent_code) |
|
|
| |
| print(f"Fetching questions from: {questions_url}") |
| try: |
| response = requests.get(questions_url, timeout=15) |
| response.raise_for_status() |
| questions_data = response.json() |
| if not questions_data: |
| print("Fetched questions list is empty.") |
| return "Fetched questions list is empty or invalid format.", None |
| print(f"Fetched {len(questions_data)} questions.") |
| except requests.exceptions.RequestException as e: |
| print(f"Error fetching questions: {e}") |
| return f"Error fetching questions: {e}", None |
| except requests.exceptions.JSONDecodeError as e: |
| print(f"Error decoding JSON response from questions endpoint: {e}") |
| print(f"Response text: {response.text[:500]}") |
| return f"Error decoding server response for questions: {e}", None |
| except Exception as e: |
| print(f"An unexpected error occurred fetching questions: {e}") |
| return f"An unexpected error occurred fetching questions: {e}", None |
|
|
| |
| results_log = [] |
| answers_payload = [] |
| print(f"Running agent on {len(questions_data)} questions...") |
| for item in questions_data: |
| task_id = item.get("task_id") |
| question_text = item.get("question") |
| if not task_id or question_text is None: |
| print(f"Skipping item with missing task_id or question: {item}") |
| continue |
| try: |
| submitted_answer = agent(question_text) |
| answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer}) |
| results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer}) |
| except Exception as e: |
| print(f"Error running agent on task {task_id}: {e}") |
| results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"}) |
|
|
| if not answers_payload: |
| print("Agent did not produce any answers to submit.") |
| return "Agent did not produce any answers to submit.", pd.DataFrame(results_log) |
|
|
| |
| submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload} |
| status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..." |
| print(status_update) |
|
|
| |
| print(f"Submitting {len(answers_payload)} answers to: {submit_url}") |
| try: |
| response = requests.post(submit_url, json=submission_data, timeout=60) |
| response.raise_for_status() |
| result_data = response.json() |
| final_status = ( |
| f"Submission Successful!\n" |
| f"User: {result_data.get('username')}\n" |
| f"Overall Score: {result_data.get('score', 'N/A')}% " |
| f"({result_data.get('correct_count', '?')}/{result_data.get('total_attempted', '?')} correct)\n" |
| f"Message: {result_data.get('message', 'No message received.')}" |
| ) |
| print("Submission successful.") |
| results_df = pd.DataFrame(results_log) |
| return final_status, results_df |
| except requests.exceptions.HTTPError as e: |
| error_detail = f"Server responded with status {e.response.status_code}." |
| try: |
| error_json = e.response.json() |
| error_detail += f" Detail: {error_json.get('detail', e.response.text)}" |
| except requests.exceptions.JSONDecodeError: |
| error_detail += f" Response: {e.response.text[:500]}" |
| status_message = f"Submission Failed: {error_detail}" |
| print(status_message) |
| results_df = pd.DataFrame(results_log) |
| return status_message, results_df |
| except requests.exceptions.Timeout: |
| status_message = "Submission Failed: The request timed out." |
| print(status_message) |
| results_df = pd.DataFrame(results_log) |
| return status_message, results_df |
| except requests.exceptions.RequestException as e: |
| status_message = f"Submission Failed: Network error - {e}" |
| print(status_message) |
| results_df = pd.DataFrame(results_log) |
| return status_message, results_df |
| except Exception as e: |
| status_message = f"An unexpected error occurred during submission: {e}" |
| print(status_message) |
| results_df = pd.DataFrame(results_log) |
| return status_message, results_df |
|
|
|
|
| |
| with gr.Blocks() as demo: |
| gr.Markdown("# Basic Agent Evaluation Runner") |
| gr.Markdown( |
| """ |
| **Instructions:** |
| |
| 1. Please clone this space, then modify the code to define your agent's logic, the tools, the necessary packages, etc ... |
| 2. Log in to your Hugging Face account using the button below. This uses your HF username for submission. |
| 3. Click 'Run Evaluation & Submit All Answers' to fetch questions, run your agent, submit answers, and see the score. |
| |
| --- |
| **Disclaimers:** |
| Once clicking on the "submit button, it can take quite some time ( this is the time for the agent to go through all the questions). |
| This space provides a basic setup and is intentionally sub-optimal to encourage you to develop your own, more robust solution. For instance for the delay process of the submit button, a solution could be to cache the answers and submit in a seperate action or even to answer the questions in async. |
| """ |
| ) |
|
|
| gr.LoginButton() |
|
|
| with gr.Row(): |
| question_box = gr.Textbox(label="Random Question", lines=4) |
| answer_box = gr.Textbox(label="Agent Answer", lines=2) |
| result_box = gr.Textbox(label="Evaluation Result", lines=1) |
|
|
| test_button = gr.Button("🔁 Answer Random Question & Evaluate") |
| test_button.click(fn=test_random_question, inputs=[], outputs=[question_box, answer_box, result_box]) |
|
|
|
|
| run_button = gr.Button("Run Evaluation & Submit All Answers") |
|
|
| status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False) |
| |
| results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True) |
|
|
| run_button.click( |
| fn=run_and_submit_all, |
| outputs=[status_output, results_table] |
| ) |
|
|
| if __name__ == "__main__": |
| print("\n" + "-"*30 + " App Starting " + "-"*30) |
| |
| space_host_startup = os.getenv("SPACE_HOST") |
| space_id_startup = os.getenv("SPACE_ID") |
|
|
| if space_host_startup: |
| print(f"✅ SPACE_HOST found: {space_host_startup}") |
| print(f" Runtime URL should be: https://{space_host_startup}.hf.space") |
| else: |
| print("ℹ️ SPACE_HOST environment variable not found (running locally?).") |
|
|
| if space_id_startup: |
| print(f"✅ SPACE_ID found: {space_id_startup}") |
| print(f" Repo URL: https://huggingface.co/spaces/{space_id_startup}") |
| print(f" Repo Tree URL: https://huggingface.co/spaces/{space_id_startup}/tree/main") |
| else: |
| print("ℹ️ SPACE_ID environment variable not found (running locally?). Repo URL cannot be determined.") |
|
|
| print("-"*(60 + len(" App Starting ")) + "\n") |
|
|
| print("Launching Gradio Interface for Basic Agent Evaluation...") |
| demo.launch(debug=True, share=False) |