Final_Assignment_Temp

Sleeping

App Files Files Community

AlanRocha commited on 14 days ago

Commit

aa7dde1

verified ·

1 Parent(s): 9950dc7

Update app.py

Browse files

Files changed (1) hide show

app.py +220 -259

app.py CHANGED Viewed

@@ -1,36 +1,53 @@
 import os
 import gradio as gr
 import requests
-import inspect
 import pandas as pd
 import tempfile
 import threading
 import queue
-from smolagents import CodeAgent, InferenceClientModel, LiteLLMModel, WebSearchTool, tool
-# (Keep Constants as is)
 # --- Constants ---
 DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
 # --- Custom tools for reading task attachments ---
 def _download_task_file(task_id: str) -> str:
-    """Internal helper: downloads the file attached to a task_id and saves
-    it to a temp folder, returning the local file path (or '' if none)."""
     url = f"{DEFAULT_API_URL}/files/{task_id}"
     try:
         response = requests.get(url, timeout=8)
         if response.status_code != 200:
             return ""
-        # Try to get a filename from the Content-Disposition header
         cd = response.headers.get("content-disposition", "")
         filename = task_id
         if "filename=" in cd:
             filename = cd.split("filename=")[-1].strip('"; ')
         else:
-            # Guess an extension from content-type
             ctype = response.headers.get("content-type", "")
             if "spreadsheet" in ctype or "excel" in ctype:
                 filename = f"{task_id}.xlsx"
@@ -63,7 +80,7 @@ def download_task_file(task_id: str) -> str:
     Returns:
         The local file path where the file was saved, or 'NO_FILE_AVAILABLE' if there
-        is no file for this task_id (in that case, do not retry - use web_search instead).
     """
     result = _download_task_file(task_id)
     return result if result else "NO_FILE_AVAILABLE"
@@ -71,8 +88,7 @@ def download_task_file(task_id: str) -> str:
 @tool
 def read_excel_file(file_path: str) -> str:
-    """Reads an Excel (.xlsx/.xls) file and returns its content as readable text
-    (one table per sheet). Use this after downloading the file with download_task_file.
     Args:
         file_path: Local path to the Excel file.
@@ -143,94 +159,108 @@ def transcribe_audio_file(file_path: str) -> str:
         return f"Error transcribing audio file: {e}"
-# --- Basic Agent Definition ---
-# ----- THIS IS WERE YOU CAN BUILD WHAT YOU WANT ------
 class BasicAgent:
     """
-    A real agent built with smolagents.
-    Uses a free Hugging Face hosted model, web search, and a set of file
-    tools (Excel/CSV/text/audio) to handle GAIA-style benchmark questions
-    that come with an attachment.
     """
     def __init__(self):
         print("BasicAgent initializing...")
-        gemini_key = os.getenv("GEMINI_API_KEY")
-        groq_key = os.getenv("GROQ_API_KEY")
-        cerebras_key = os.getenv("CEREBRAS_API_KEY")
-        # Build a router that tries multiple free providers in order and
-        # automatically fails over to the next one if a call errors out
-        # (rate limit, quota exceeded, timeout, etc). This means a single
-        # exhausted free tier no longer kills the whole 20-question run -
-        # the router just moves to the next provider for the NEXT call.
         model_list = []
         if cerebras_key:
-            # Cerebras free tier: up to 1M tokens/day, fast, no waitlist -
-            # the most generous free option, so it goes first.
             model_list.append({
                 "model_name": "agent-model",
                 "litellm_params": {
-                    "model": "cerebras/gpt-oss-120b",
                     "api_key": cerebras_key,
                     "num_retries": 0,
-                    "timeout": 8,
                 },
             })
-        if groq_key:
-            # Groq: generous free tier (100k tokens/day), very fast LPUs.
             model_list.append({
                 "model_name": "agent-model",
                 "litellm_params": {
-                    "model": "groq/llama-3.3-70b-versatile",
-                    "api_key": groq_key,
                     "num_retries": 0,
-                    "timeout": 8,
                 },
             })
         if gemini_key:
-            # Gemini: good quality, but this account's free tier is capped
-            # at only 20 requests/day, so it's last among the paid-key options.
             model_list.append({
                 "model_name": "agent-model",
                 "litellm_params": {
                     "model": "gemini/gemini-2.5-flash",
                     "api_key": gemini_key,
                     "num_retries": 0,
-                    "timeout": 8,
                 },
             })
         if model_list:
             from smolagents import LiteLLMRouterModel
-            provider_names = []
-            if cerebras_key:
-                provider_names.append("Cerebras")
-            if groq_key:
-                provider_names.append("Groq")
-            if gemini_key:
-                provider_names.append("Gemini")
-            print(f"Using router with fallback chain: {' -> '.join(provider_names)}")
             self.model = LiteLLMRouterModel(
                 model_id="agent-model",
                 model_list=model_list,
                 client_kwargs={
-                    "num_retries": 0,        # router-level: also fail fast
-                    "timeout": 20,
                     "routing_strategy": "simple-shuffle",
                 },
             )
         else:
-            # Final fallback: free model hosted by Hugging Face Inference Providers.
-            print("No CEREBRAS/GROQ/GEMINI API key set - falling back to HF Inference Providers.")
             self.model = InferenceClientModel(
-                model_id="Qwen/Qwen2.5-Coder-32B-Instruct",
             )
-        from smolagents import PythonInterpreterTool
-        self.agent = CodeAgent(
             tools=[
                 WebSearchTool(),
                 download_task_file,
@@ -238,291 +268,222 @@ class BasicAgent:
                 read_csv_file,
                 read_text_file,
                 transcribe_audio_file,
-                PythonInterpreterTool(),
             ],
             model=self.model,
-            add_base_tools=False,  # we add tools manually below to EXCLUDE visit_webpage,
-                                    # which has caused 150-800 second hangs in testing
-            additional_authorized_imports=[
-                "pandas", "numpy", "json", "re", "math", "datetime",
-                "openpyxl", "io", "csv",
-            ],
-            max_steps=3,   # keep this LOW for speed - we only need 30%, not perfection
         )
-        # Extra safety net: if visit_webpage somehow still ended up in the
-        # toolbox (e.g. via a future smolagents default change), remove it.
-        if "visit_webpage" in self.agent.tools:
-            del self.agent.tools["visit_webpage"]
         print("BasicAgent initialized.")
     def __call__(self, question: str, task_id: str = "") -> str:
-        print(f"Agent received question (first 50 chars): {question[:50]}...")
-        # Some GAIA questions are written backwards as a "riddle" test.
-        # Detect this and flip it back before sending to the model.
-        reversed_hint = ""
-        if question.strip().endswith(".") and question.strip()[:1].islower():
-            # crude heuristic: try reversing and see if it reads like English
-            flipped = question.strip()[::-1]
-            if flipped[:1].isupper() or flipped.split(" ")[0].isalpha():
-                reversed_hint = (
-                    f"\n\nNote: this question may be written backwards. "
-                    f"Reversed, it reads: {flipped}"
-                )
-        # Strong instruction to keep answers in the exact-match format
-        # the GAIA benchmark expects: no "FINAL ANSWER" prefix, no extra
-        # explanation, just the bare answer.
-        instructions = (
-            "You are a general AI assistant answering a benchmark question. "
-            "You have a STRICT step budget (max 3 steps) - be fast and efficient, "
-            "do not waste steps retrying things that already failed.\n\n"
-            "RULES TO SAVE STEPS AND TIME:\n"
-            "- PREFER web_search over visit_webpage in almost all cases - it is faster "
-            "and more reliable. Only use visit_webpage if web_search snippets are not "
-            "enough AND the URL is not youtube.com/youtu.be.\n"
-            "- NEVER call visit_webpage on a youtube.com/youtu.be URL - it always "
-            "fails with a connection error. For video questions, only use web_search "
-            "to find what others have already said about the video content.\n"
-            "- If download_task_file returns 'NO_FILE_AVAILABLE', do NOT call it "
-            "again - immediately move on to web_search instead.\n"
-            "- If visit_webpage returns a 403 or connection error, do NOT retry the "
-            "same URL - immediately try web_search instead.\n"
-            "- Answer in as few steps as possible - ideally in just 1 step if you "
-            "already know the answer or can compute it directly. Do not over-verify.\n\n"
-            f"The task_id for this question is '{task_id}'. If the question "
-            "mentions an attached file (Excel, CSV, audio, image, code, etc.), "
-            "call download_task_file('" + task_id + "') ONCE first to get its local "
-            "path, then use the matching reading tool (read_excel_file, "
-            "read_csv_file, read_text_file, or transcribe_audio_file) on that path.\n\n"
-            "Report your thoughts, then finish with the answer. "
-            "Your final output must be ONLY the answer itself: "
-            "no explanations, no extra words, no 'FINAL ANSWER' prefix. "
-            "If the answer is a number, write only the number (no units unless "
-            "explicitly requested). If it's a string, give the minimal exact phrase "
-            "requested, avoiding articles and abbreviations unless asked otherwise. "
-            "If it's a list, give a comma separated list following the same rules."
-            f"{reversed_hint}\n\n"
             f"Question: {question}"
         )
-        # HARD TIMEOUT: run the agent in a background thread and give up after
-        # a fixed number of seconds no matter what is happening internally
-        # (rate limit waits, hanging network calls, retries the library does
-        # on its own, etc). This guarantees the whole 20-question run can
-        # never stall for minutes/hours on a single question.
-        PER_QUESTION_TIMEOUT = 300  # seconds
         result_queue: "queue.Queue" = queue.Queue()
-        def _run_agent():
             try:
-                r = self.agent.run(instructions)
                 result_queue.put(("ok", r))
             except Exception as exc:
                 result_queue.put(("error", exc))
-        worker = threading.Thread(target=_run_agent, daemon=True)
         worker.start()
         worker.join(timeout=PER_QUESTION_TIMEOUT)
         if worker.is_alive():
-            # Still running after the deadline - give up on this question and
-            # move on. The thread is daemonized so it won't block process exit.
-            print(
-                f"Agent exceeded {PER_QUESTION_TIMEOUT}s hard timeout - "
-                "abandoning this question and moving to the next one."
-            )
-            answer = "I don't know."
         else:
-            try:
-                status, payload = result_queue.get_nowait()
-            except Exception:
-                status, payload = "error", "no result produced"
-            if status == "ok":
-                answer = str(payload).strip()
-            else:
-                print(f"Agent error while answering: {payload}")
-                answer = "I don't know."
-        print(f"Agent returning answer: {answer}")
         return answer
-def run_and_submit_all( profile: gr.OAuthProfile | None):
-    """
-    Fetches all questions, runs the BasicAgent on them, submits all answers,
-    and displays the results.
-    """
-    # --- Determine HF Space Runtime URL and Repo URL ---
-    space_id = os.getenv("SPACE_ID") # Get the SPACE_ID for sending link to the code
-    if profile:
-        username= f"{profile.username}"
-        print(f"User logged in: {username}")
-    else:
-        print("User not logged in.")
-        return "Please Login to Hugging Face with the button.", None
-    api_url = DEFAULT_API_URL
     questions_url = f"{api_url}/questions"
-    submit_url = f"{api_url}/submit"
-    # 1. Instantiate Agent ( modify this part to create your agent)
     try:
         agent = BasicAgent()
     except Exception as e:
-        print(f"Error instantiating agent: {e}")
         return f"Error initializing agent: {e}", None
-    # In the case of an app running as a hugging Face space, this link points toward your codebase ( usefull for others so please keep it public)
-    agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main"
-    print(agent_code)
-    # 2. Fetch Questions
     print(f"Fetching questions from: {questions_url}")
     try:
-        response = requests.get(questions_url, timeout=15)
-        response.raise_for_status()
-        questions_data = response.json()
         if not questions_data:
-             print("Fetched questions list is empty.")
-             return "Fetched questions list is empty or invalid format.", None
         print(f"Fetched {len(questions_data)} questions.")
-    except requests.exceptions.RequestException as e:
-        print(f"Error fetching questions: {e}")
-        return f"Error fetching questions: {e}", None
-    except requests.exceptions.JSONDecodeError as e:
-         print(f"Error decoding JSON response from questions endpoint: {e}")
-         print(f"Response text: {response.text[:500]}")
-         return f"Error decoding server response for questions: {e}", None
     except Exception as e:
-        print(f"An unexpected error occurred fetching questions: {e}")
-        return f"An unexpected error occurred fetching questions: {e}", None
-    # 3. Run your Agent
-    results_log = []
     answers_payload = []
-    print(f"Running agent on {len(questions_data)} questions...")
     for item in questions_data:
-        task_id = item.get("task_id")
         question_text = item.get("question")
         if not task_id or question_text is None:
-            print(f"Skipping item with missing task_id or question: {item}")
             continue
-        try:
-            submitted_answer = agent(question_text, task_id=task_id)
-            answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer})
-            results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer})
-        except Exception as e:
-             print(f"Error running agent on task {task_id}: {e}")
-             results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"})
     if not answers_payload:
-        print("Agent did not produce any answers to submit.")
-        return "Agent did not produce any answers to submit.", pd.DataFrame(results_log)
-    # 4. Prepare Submission
-    submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload}
-    status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..."
-    print(status_update)
-    # 5. Submit
-    print(f"Submitting {len(answers_payload)} answers to: {submit_url}")
     try:
-        response = requests.post(submit_url, json=submission_data, timeout=60)
-        response.raise_for_status()
-        result_data = response.json()
         final_status = (
             f"Submission Successful!\n"
             f"User: {result_data.get('username')}\n"
-            f"Overall Score: {result_data.get('score', 'N/A')}% "
             f"({result_data.get('correct_count', '?')}/{result_data.get('total_attempted', '?')} correct)\n"
             f"Message: {result_data.get('message', 'No message received.')}"
         )
         print("Submission successful.")
-        results_df = pd.DataFrame(results_log)
-        return final_status, results_df
     except requests.exceptions.HTTPError as e:
-        error_detail = f"Server responded with status {e.response.status_code}."
         try:
-            error_json = e.response.json()
-            error_detail += f" Detail: {error_json.get('detail', e.response.text)}"
-        except requests.exceptions.JSONDecodeError:
-            error_detail += f" Response: {e.response.text[:500]}"
-        status_message = f"Submission Failed: {error_detail}"
-        print(status_message)
-        results_df = pd.DataFrame(results_log)
-        return status_message, results_df
-    except requests.exceptions.Timeout:
-        status_message = "Submission Failed: The request timed out."
-        print(status_message)
-        results_df = pd.DataFrame(results_log)
-        return status_message, results_df
-    except requests.exceptions.RequestException as e:
-        status_message = f"Submission Failed: Network error - {e}"
-        print(status_message)
-        results_df = pd.DataFrame(results_log)
-        return status_message, results_df
     except Exception as e:
-        status_message = f"An unexpected error occurred during submission: {e}"
-        print(status_message)
-        results_df = pd.DataFrame(results_log)
-        return status_message, results_df
-# --- Build Gradio Interface using Blocks ---
 with gr.Blocks() as demo:
-    gr.Markdown("# Basic Agent Evaluation Runner")
     gr.Markdown(
         """
         **Instructions:**
-        1.  Please clone this space, then modify the code to define your agent's logic, the tools, the necessary packages, etc ...
-        2.  Log in to your Hugging Face account using the button below. This uses your HF username for submission.
-        3.  Click 'Run Evaluation & Submit All Answers' to fetch questions, run your agent, submit answers, and see the score.
-        ---
-        **Disclaimers:**
-        Once clicking on the "submit button, it can take quite some time ( this is the time for the agent to go through all the questions).
-        This space provides a basic setup and is intentionally sub-optimal to encourage you to develop your own, more robust solution. For instance for the delay process of the submit button, a solution could be to cache the answers and submit in a seperate action or even to answer the questions in async.
         """
     )
     gr.LoginButton()
-    run_button = gr.Button("Run Evaluation & Submit All Answers")
-    status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False)
-    # Removed max_rows=10 from DataFrame constructor
-    results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True)
-    run_button.click(
-        fn=run_and_submit_all,
-        outputs=[status_output, results_table]
-    )
 if __name__ == "__main__":
-    print("\n" + "-"*30 + " App Starting " + "-"*30)
-    # Check for SPACE_HOST and SPACE_ID at startup for information
-    space_host_startup = os.getenv("SPACE_HOST")
-    space_id_startup = os.getenv("SPACE_ID") # Get SPACE_ID at startup
-    if space_host_startup:
-        print(f"✅ SPACE_HOST found: {space_host_startup}")
-        print(f"   Runtime URL should be: https://{space_host_startup}.hf.space")
-    else:
-        print("ℹ️  SPACE_HOST environment variable not found (running locally?).")
-    if space_id_startup: # Print repo URLs if SPACE_ID is found
-        print(f"✅ SPACE_ID found: {space_id_startup}")
-        print(f"   Repo URL: https://huggingface.co/spaces/{space_id_startup}")
-        print(f"   Repo Tree URL: https://huggingface.co/spaces/{space_id_startup}/tree/main")
     else:
-        print("ℹ️  SPACE_ID environment variable not found (running locally?). Repo URL cannot be determined.")
-    print("-"*(60 + len(" App Starting ")) + "\n")
-    print("Launching Gradio Interface for Basic Agent Evaluation...")
     demo.launch(debug=True, share=False, ssr_mode=False)

 import os
+import json
+import time
 import gradio as gr
 import requests
 import pandas as pd
 import tempfile
 import threading
 import queue
+from smolagents import ToolCallingAgent, InferenceClientModel, LiteLLMModel, WebSearchTool, tool
 # --- Constants ---
 DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
+CACHE_FILE = "/tmp/gaia_answers_cache.json"
+# --- Cache helpers ---
+def load_cache() -> dict:
+    if os.path.exists(CACHE_FILE):
+        try:
+            with open(CACHE_FILE, "r") as f:
+                return json.load(f)
+        except Exception:
+            return {}
+    return {}
+def save_cache(cache: dict):
+    try:
+        with open(CACHE_FILE, "w") as f:
+            json.dump(cache, f)
+    except Exception as e:
+        print(f"Warning: could not save cache: {e}")
 # --- Custom tools for reading task attachments ---
 def _download_task_file(task_id: str) -> str:
     url = f"{DEFAULT_API_URL}/files/{task_id}"
     try:
         response = requests.get(url, timeout=8)
         if response.status_code != 200:
             return ""
         cd = response.headers.get("content-disposition", "")
         filename = task_id
         if "filename=" in cd:
             filename = cd.split("filename=")[-1].strip('"; ')
         else:
             ctype = response.headers.get("content-type", "")
             if "spreadsheet" in ctype or "excel" in ctype:
                 filename = f"{task_id}.xlsx"
     Returns:
         The local file path where the file was saved, or 'NO_FILE_AVAILABLE' if there
+        is no file for this task_id.
     """
     result = _download_task_file(task_id)
     return result if result else "NO_FILE_AVAILABLE"
 @tool
 def read_excel_file(file_path: str) -> str:
+    """Reads an Excel (.xlsx/.xls) file and returns its content as readable text.
     Args:
         file_path: Local path to the Excel file.
         return f"Error transcribing audio file: {e}"
+# --- Agent ---
 class BasicAgent:
     """
+    Token-efficient agent for the GAIA benchmark.
+    Key optimizations vs the original:
+    - ToolCallingAgent instead of CodeAgent  → ~40% fewer tokens per step
+    - Small/fast model first (Groq llama-3.1-8b-instant, free tier)
+    - Lean prompt (~80 tokens instead of ~400)
+    - Per-run answer cache so re-runs never re-spend tokens on answered questions
+    - Hard 120 s timeout per question (down from 300 s)
     """
     def __init__(self):
         print("BasicAgent initializing...")
+        groq_key      = os.getenv("GROQ_API_KEY")
+        cerebras_key  = os.getenv("CEREBRAS_API_KEY")
+        gemini_key    = os.getenv("GEMINI_API_KEY")
+        anthropic_key = os.getenv("ANTHROPIC_API_KEY")
+        # Build priority list: cheapest/fastest first.
         model_list = []
+        if groq_key:
+            # Groq free tier — 70B first for quality, 8B as fallback when rate limited
+            model_list.append({
+                "model_name": "agent-model",
+                "litellm_params": {
+                    "model": "groq/llama-3.3-70b-versatile",
+                    "api_key": groq_key,
+                    "num_retries": 0,
+                    "timeout": 20,
+                },
+            })
+            model_list.append({
+                "model_name": "agent-model",
+                "litellm_params": {
+                    "model": "groq/llama-3.1-8b-instant",
+                    "api_key": groq_key,
+                    "num_retries": 0,
+                    "timeout": 15,
+                },
+            })
         if cerebras_key:
             model_list.append({
                 "model_name": "agent-model",
                 "litellm_params": {
+                    "model": "cerebras/llama3.1-8b",   # free, very fast
                     "api_key": cerebras_key,
                     "num_retries": 0,
+                    "timeout": 15,
                 },
             })
+        if anthropic_key:
+            # Haiku is Anthropic's cheapest model — ~25x cheaper than Sonnet.
             model_list.append({
                 "model_name": "agent-model",
                 "litellm_params": {
+                    "model": "anthropic/claude-haiku-4-5-20251001",
+                    "api_key": anthropic_key,
                     "num_retries": 0,
+                    "timeout": 20,
                 },
             })
         if gemini_key:
+            # gemini-2.5-flash: free tier, 15 RPM, 1500 RPD — gemini-2.0-flash was deprecated Jun 2026
             model_list.append({
                 "model_name": "agent-model",
                 "litellm_params": {
                     "model": "gemini/gemini-2.5-flash",
                     "api_key": gemini_key,
                     "num_retries": 0,
+                    "timeout": 20,
                 },
             })
         if model_list:
             from smolagents import LiteLLMRouterModel
+            print(f"Router with {len(model_list)} model slots configured.")
             self.model = LiteLLMRouterModel(
                 model_id="agent-model",
                 model_list=model_list,
                 client_kwargs={
+                    "num_retries": 0,
+                    "timeout": 30,
                     "routing_strategy": "simple-shuffle",
                 },
             )
         else:
+            print("No API keys found — falling back to HF Inference Providers (Qwen 7B).")
             self.model = InferenceClientModel(
+                model_id="Qwen/Qwen2.5-7B-Instruct",  # smaller than 32B
             )
+        # ToolCallingAgent: generates only tool-call JSON, not full Python code.
+        # This alone cuts token usage by ~40% compared to CodeAgent.
+        self.agent = ToolCallingAgent(
             tools=[
                 WebSearchTool(),
                 download_task_file,
                 read_csv_file,
                 read_text_file,
                 transcribe_audio_file,
             ],
             model=self.model,
+            max_steps=4,
         )
         print("BasicAgent initialized.")
     def __call__(self, question: str, task_id: str = "") -> str:
+        print(f"Agent received question (first 60 chars): {question[:60]}...")
+        # Lean prompt — every token here is multiplied by max_steps calls.
+        prompt = (
+            "You are a precise AI assistant. Answer the question below with ONLY "
+            "the bare answer (no explanation, no preamble, no 'FINAL ANSWER' prefix). "
+            "Numbers: digits only unless units explicitly requested. "
+            "Strings: minimal exact phrase. Lists: comma-separated.\n"
+            f"task_id='{task_id}' — if the question mentions an attached file, "
+            "call download_task_file ONCE first, then the matching read tool.\n"
+            "Use web_search for factual lookups; never visit youtube URLs.\n\n"
             f"Question: {question}"
         )
+        PER_QUESTION_TIMEOUT = 120  # seconds — tighter than original 300 s
         result_queue: "queue.Queue" = queue.Queue()
+        def _run():
             try:
+                r = self.agent.run(prompt)
                 result_queue.put(("ok", r))
             except Exception as exc:
                 result_queue.put(("error", exc))
+        worker = threading.Thread(target=_run, daemon=True)
         worker.start()
         worker.join(timeout=PER_QUESTION_TIMEOUT)
         if worker.is_alive():
+            print(f"Timeout after {PER_QUESTION_TIMEOUT}s — skipping question.")
+            return "I don't know."
+        try:
+            status, payload = result_queue.get_nowait()
+        except Exception:
+            return "I don't know."
+        if status == "ok":
+            answer = str(payload).strip()
         else:
+            print(f"Agent error: {payload}")
+            answer = "I don't know."
+        print(f"Agent answer: {answer}")
         return answer
+# --- Main evaluation runner ---
+def run_and_submit_all(profile: gr.OAuthProfile | None):
+    space_id = os.getenv("SPACE_ID")
+    if not profile:
+        return "Please log in to Hugging Face first.", None
+    username = profile.username
+    print(f"Logged in as: {username}")
+    api_url       = DEFAULT_API_URL
     questions_url = f"{api_url}/questions"
+    submit_url    = f"{api_url}/submit"
+    # --- Instantiate agent ---
     try:
         agent = BasicAgent()
     except Exception as e:
         return f"Error initializing agent: {e}", None
+    agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main" if space_id else "unknown"
+    print(f"Agent code URL: {agent_code}")
+    # --- Fetch questions ---
     print(f"Fetching questions from: {questions_url}")
     try:
+        resp = requests.get(questions_url, timeout=15)
+        resp.raise_for_status()
+        questions_data = resp.json()
         if not questions_data:
+            return "Fetched questions list is empty.", None
         print(f"Fetched {len(questions_data)} questions.")
     except Exception as e:
+        return f"Error fetching questions: {e}", None
+    # --- Load cache (avoids re-spending tokens on already-answered questions) ---
+    cache = load_cache()
+    print(f"Cache loaded: {len(cache)} previously answered questions.")
+    # --- Run agent ---
+    results_log    = []
     answers_payload = []
     for item in questions_data:
+        task_id       = item.get("task_id")
         question_text = item.get("question")
         if not task_id or question_text is None:
+            print(f"Skipping malformed item: {item}")
             continue
+        if task_id in cache:
+            submitted_answer = cache[task_id]
+            print(f"[CACHE HIT] task_id={task_id} → {submitted_answer}")
+        else:
+            try:
+                submitted_answer = agent(question_text, task_id=task_id)
+            except Exception as e:
+                print(f"Error on task {task_id}: {e}")
+                submitted_answer = "I don't know."
+            cache[task_id] = submitted_answer
+            save_cache(cache)  # persist after every answer so crashes don't lose progress
+            # Rate limit guard: 5s between questions keeps us under 12 req/min,
+            # safely below Gemini free tier's 15 RPM and Groq's burst limits.
+            print("Waiting 5s before next question (rate limit guard)...")
+            time.sleep(5)
+        answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer})
+        results_log.append({
+            "Task ID": task_id,
+            "Question": question_text,
+            "Submitted Answer": submitted_answer,
+        })
     if not answers_payload:
+        return "Agent produced no answers.", pd.DataFrame(results_log)
+    # --- Submit ---
+    submission_data = {
+        "username": username.strip(),
+        "agent_code": agent_code,
+        "answers": answers_payload,
+    }
+    print(f"Submitting {len(answers_payload)} answers...")
     try:
+        resp = requests.post(submit_url, json=submission_data, timeout=60)
+        resp.raise_for_status()
+        result_data = resp.json()
         final_status = (
             f"Submission Successful!\n"
             f"User: {result_data.get('username')}\n"
+            f"Score: {result_data.get('score', 'N/A')}% "
             f"({result_data.get('correct_count', '?')}/{result_data.get('total_attempted', '?')} correct)\n"
             f"Message: {result_data.get('message', 'No message received.')}"
         )
         print("Submission successful.")
+        return final_status, pd.DataFrame(results_log)
     except requests.exceptions.HTTPError as e:
+        detail = f"HTTP {e.response.status_code}"
         try:
+            detail += f" — {e.response.json().get('detail', e.response.text)}"
+        except Exception:
+            detail += f" — {e.response.text[:300]}"
+        return f"Submission failed: {detail}", pd.DataFrame(results_log)
     except Exception as e:
+        return f"Submission failed: {e}", pd.DataFrame(results_log)
+# --- Gradio UI ---
 with gr.Blocks() as demo:
+    gr.Markdown("# GAIA Agent Evaluation Runner")
     gr.Markdown(
         """
         **Instructions:**
+        1. Clone this Space and add your API keys as Secrets (`GROQ_API_KEY`, `CEREBRAS_API_KEY`, `ANTHROPIC_API_KEY`, or `GEMINI_API_KEY`).
+        2. Log in with your Hugging Face account below.
+        3. Click **Run Evaluation & Submit All Answers**.
+        **Token-saving features in this version:**
+        - `ToolCallingAgent` instead of `CodeAgent` (~40% fewer tokens/step)
+        - Groq `llama-3.3-70b-versatile` as primary (free tier, 100k tokens/day)
+        - Groq `llama-3.1-8b-instant` as first fallback (free, very fast)
+        - Gemini `gemini-2.5-flash` as second fallback (free, 1500 req/day)
+        - Lean prompt (~80 tokens vs ~400 in the original)
+        - Answer cache: re-runs never re-spend tokens on already-answered questions
+        - 5s sleep between questions to avoid 429 rate limit errors
+        - 120s hard timeout per question
         """
     )
     gr.LoginButton()
+    run_button   = gr.Button("Run Evaluation & Submit All Answers")
+    status_output = gr.Textbox(label="Status / Result", lines=6, interactive=False)
+    results_table = gr.DataFrame(label="Questions and Answers", wrap=True)
+    run_button.click(fn=run_and_submit_all, outputs=[status_output, results_table])
 if __name__ == "__main__":
+    print("\n" + "-" * 30 + " App Starting " + "-" * 30)
+    space_host = os.getenv("SPACE_HOST")
+    space_id   = os.getenv("SPACE_ID")
+    if space_host:
+        print(f"✅ SPACE_HOST: {space_host}")
     else:
+        print("ℹ️  SPACE_HOST not set (running locally?).")
+    if space_id:
+        print(f"✅ SPACE_ID: {space_id}")
+        print(f"   Repo: https://huggingface.co/spaces/{space_id}/tree/main")
+    else:
+        print("ℹ️  SPACE_ID not set (running locally?).")
+    print("-" * (60 + len(" App Starting ")) + "\n")
+    print("Launching Gradio interface...")
     demo.launch(debug=True, share=False, ssr_mode=False)