Spaces:
Running on Zero
Running on Zero
File size: 30,856 Bytes
06094f8 10e9b7d ef2c544 06094f8 ef2c544 06094f8 10e9b7d eccf8e4 3c4371f 10e9b7d ef2c544 06094f8 e80aab9 3db6293 e80aab9 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 31243f4 06094f8 991cedd ef2c544 c489c3a ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 ef2c544 06094f8 31243f4 06094f8 31243f4 06094f8 3c4371f 7e4a06b 06094f8 3c4371f 7e4a06b 3c4371f 7d65c66 3c4371f 7e4a06b 31243f4 e80aab9 31243f4 06094f8 31243f4 3c4371f 31243f4 06094f8 36ed51a c1fd3d2 3c4371f 31243f4 eccf8e4 31243f4 7d65c66 31243f4 06094f8 31243f4 e80aab9 31243f4 3c4371f 06094f8 7d65c66 31243f4 e80aab9 7d65c66 3c4371f 31243f4 06094f8 31243f4 06094f8 31243f4 ef2c544 7d65c66 31243f4 ef2c544 06094f8 31243f4 3c4371f 31243f4 7d65c66 3c4371f 31243f4 e80aab9 31243f4 e80aab9 7d65c66 e80aab9 31243f4 e80aab9 3c4371f e80aab9 31243f4 e80aab9 3c4371f e80aab9 3c4371f e80aab9 7d65c66 3c4371f 31243f4 7d65c66 31243f4 3c4371f e80aab9 31243f4 7d65c66 31243f4 e80aab9 31243f4 0ee0419 e514fd7 06094f8 e514fd7 06094f8 e514fd7 e80aab9 7e4a06b e80aab9 31243f4 e80aab9 9088b99 7d65c66 e80aab9 31243f4 e80aab9 ef2c544 e80aab9 06094f8 3c4371f 06094f8 7d65c66 3c4371f ef2c544 3c4371f 06094f8 7d65c66 06094f8 7d65c66 06094f8 7d65c66 06094f8 3c4371f 31243f4 3c4371f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 | import spaces # must be the first import — required by ZeroGPU Spaces
import os
import re
import tempfile
import time
from pathlib import Path
from typing import Optional
import gradio as gr
import requests
import pandas as pd
from smolagents import CodeAgent, InferenceClientModel, OpenAIServerModel
from smolagents.default_tools import DuckDuckGoSearchTool, VisitWebpageTool
# --- Constants ---
DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"
# Pinned explicitly rather than relying on InferenceClientModel's own default,
# which changes between smolagents releases. The GAIA_MODEL_ID env var still
# overrides this if a different model is ever needed.
DEFAULT_MODEL_ID = "meta-llama/Llama-3.3-70B-Instruct"
# This Space's free-tier hardware is ZeroGPU, which refuses to start unless
# at least one function is decorated with @spaces.GPU. The agent itself never
# touches a GPU (it only calls HF Inference Providers over HTTP), so this
# function exists purely to satisfy that startup check and is never called.
@spaces.GPU
def _zerogpu_startup_check():
return None
# Free provider tiers cap requests per minute (Google's Gemini free tier
# reports "limit: 20" for gemini-2.5-flash). Pacing our own calls below that
# ceiling is far cheaper than discovering it via 429s and backoff. Override
# with GAIA_RPM when moving to a tier with a different budget.
DEFAULT_REQUESTS_PER_MINUTE = 15.0
# Instructions appended to every question so the CodeAgent behaves like a
# GAIA-solving agent rather than a free-form chatbot. Kept as plain task
# framing (not a tool) to avoid adding moving parts.
TASK_RULES = """
Rules:
1. Think step by step. Break multi-part questions into sub-steps before acting.
2. Use web search for anything requiring current facts, names, dates, or
information you are not already certain of. Never invent facts or sources.
3. Use Python (you write code as your actions) for arithmetic, counting,
sorting, filtering, or any data processing rather than doing it in your head.
4. If an attachment path is given below, open and inspect it with Python
before answering — never guess its contents.
5. If web_search fails or comes back empty, do not give up on the step: retry
with a differently worded query, or call visit_webpage directly on a likely
source URL (for example
https://en.wikipedia.org/w/index.php?search=YOUR+QUERY, or the article URL
itself). Do not try to import external packages such as wikipedia,
requests-html, bs4 or googlesearch — they are not installed, and attempting
them only wastes a step.
6. Where practical, verify an important intermediate result a second way
before finalizing.
7. Call final_answer(...) with ONLY the exact value requested:
- a bare number (no commas, no units unless the question asks for units)
- a bare string (no "The answer is", no trailing period, no quotes)
- a comma-separated list only if asked for a list, in exactly the order the
question asks for; separate elements with a comma and a single space
unless the question specifies a different separator
Do not include your reasoning or the words "FINAL ANSWER" in that value.
"""
MAX_STEPS = 12
ADDITIONAL_IMPORTS = [
"pandas", "numpy", "csv", "json", "re", "math", "statistics",
"datetime", "itertools", "collections", "io", "os", "pypdf", "openpyxl",
"requests", "unicodedata", "time",
]
# Every past step's observations are replayed into every later model call, so a
# trajectory grows quadratically and eventually exceeds what a free tier will
# accept in one request — Groq returns HTTP 413 "Request too large ... Limit
# 12000, Requested 12396" and no amount of waiting fixes it, because a single
# call already exceeds the whole per-minute budget. Recent observations are what
# the agent is reasoning about; older ones only need to stay recognizable.
RECENT_STEPS_KEPT_FULL = 2
RECENT_OBSERVATION_CHARS = 8000
OLDER_OBSERVATION_CHARS = 1500
# The model's own output is replayed as an assistant message too, and reasoning
# models are verbose enough that untrimmed history alone can breach the limit.
# The code it wrote is the part worth keeping, so older steps keep only a head.
RECENT_OUTPUT_CHARS = 4000
OLDER_OUTPUT_CHARS = 1000
def trim_observations(memory_step, agent=None) -> None:
"""Step callback: shrink observations of all but the newest few steps.
Registered on ActionStep, so it runs after each action and keeps the next
request inside the provider's per-request ceiling.
"""
def _clip(text, budget):
if not isinstance(text, str) or len(text) <= budget:
return text
return (
text[:budget]
+ f"\n...[{len(text) - budget} characters truncated to stay within "
f"the model's request size limit]"
)
steps = [s for s in getattr(agent, "memory", None).steps if hasattr(s, "observations")]
for index, step in enumerate(steps):
is_recent = index >= len(steps) - RECENT_STEPS_KEPT_FULL
if step.observations:
step.observations = _clip(
step.observations,
RECENT_OBSERVATION_CHARS if is_recent else OLDER_OBSERVATION_CHARS,
)
if step.model_output:
step.model_output = _clip(
step.model_output,
RECENT_OUTPUT_CHARS if is_recent else OLDER_OUTPUT_CHARS,
)
class RetryingWebSearchTool(DuckDuckGoSearchTool):
"""Same tool the base toolkit provides, but a rate-limited or flaky
DuckDuckGo response is retried instead of burning one of the agent's
steps. Name/description/signature are inherited unchanged, so the agent's
system prompt is identical to the stock tool's."""
max_attempts = 3
backoff_seconds = 3.0
def forward(self, query: str) -> str:
last_error = None
for attempt in range(self.max_attempts):
try:
return super().forward(query)
except Exception as e:
last_error = e
print(f"web_search attempt {attempt + 1}/{self.max_attempts} failed: {e}")
if attempt < self.max_attempts - 1:
time.sleep(self.backoff_seconds * (attempt + 1))
raise RuntimeError(
f"web_search failed after {self.max_attempts} attempts: {last_error}. "
"Try visit_webpage on a likely source URL instead."
)
class _RetryOnThrottleMixin:
"""Retries transient rate-limit / server errors from the inference endpoint.
Free provider tiers throttle aggressively, and without this a single 429
aborts the whole question. Non-transient errors (401, 402 out of credits,
400 bad request) are re-raised immediately — retrying those is pointless.
Both model classes below are constructed with retry=False, which disables
smolagents' own retryer. Leaving it on nests two exponential backoffs: its
3 attempts (60s base, doubling, jittered) run inside each of our attempts,
so one throttled call can sleep for over half an hour. This layer replaces
it because it can read the provider's own retry hint instead of guessing."""
max_attempts = 6
backoff_seconds = 20.0
RETRYABLE = (
"429", "500", "502", "503", "504", "rate limit", "too many requests", "timeout",
# Some models occasionally emit a tool call where CodeAgent expects a
# code block, and the provider rejects the request outright with
# 400 "Tool choice is none, but model called a tool". It is a sampling
# artifact rather than a bad prompt, so re-asking usually succeeds.
"tool_use_failed",
)
# A per-day quota and an over-sized single request both arrive dressed as
# rate limits, but neither clears within any backoff we would sit through:
# the daily bucket refills hours later, and a request that alone exceeds the
# per-minute ceiling will be exactly as large on the next attempt. Failing
# this question immediately leaves budget and wall-clock for the rest.
FATAL = ("tokens per day", "tpd", "request too large", "413")
# Providers usually say how long to wait; obeying that beats guessing.
# Gemini phrases it "Please retry in 32.290648364s", OpenAI-style APIs
# "retry after 12 seconds".
_RETRY_HINT = re.compile(r"retry(?:\s+after)?\s+in\s+([0-9.]+)\s*s|retry after ([0-9.]+)")
# Filled in by GAIAAgent when GAIA_MODEL_ID lists more than one model.
fallback_model_ids: list = []
def generate(self, *args, **kwargs):
last_error = None
for attempt in range(self.max_attempts):
try:
return super().generate(*args, **kwargs)
except Exception as e:
message = str(e).lower()
if any(token in message for token in self.FATAL):
# Daily quotas are per-model, so a sibling model on the same
# account usually still has budget. Switching costs one call
# and rescues every remaining question; without it the run
# ends here no matter how much wall-clock is left.
if "tokens per day" in message and self.fallback_model_ids:
self.model_id = self.fallback_model_ids.pop(0)
print(
f"Daily token quota exhausted; switching to "
f"fallback model {self.model_id}"
)
continue
raise
if not any(token in message for token in self.RETRYABLE):
raise
last_error = e
# A malformed generation clears on the next sample, so re-ask
# straight away rather than serving a rate-limit-sized backoff.
if "tool_use_failed" in message:
wait = 1.0
else:
wait = self.backoff_seconds * (attempt + 1)
hint = self._RETRY_HINT.search(message)
if hint:
# +2s of slack so we come back after the window, not on its edge
wait = max(wait, float(hint.group(1) or hint.group(2)) + 2.0)
print(
f"Model call attempt {attempt + 1}/{self.max_attempts} failed "
f"({type(e).__name__}: {str(e)[:200]}); retrying in {wait:.0f}s"
)
if attempt < self.max_attempts - 1:
time.sleep(wait)
raise last_error
class RetryingInferenceClientModel(_RetryOnThrottleMixin, InferenceClientModel):
"""HF Inference Providers, with throttle retries."""
class RetryingOpenAIServerModel(_RetryOnThrottleMixin, OpenAIServerModel):
"""Any OpenAI-compatible endpoint, with throttle retries. InferenceClient
itself cannot take a model name and a base_url together, so custom
endpoints go through this class instead."""
class IdentifiedVisitWebpageTool(VisitWebpageTool):
"""The stock tool calls requests.get() with no User-Agent, so Wikipedia and
several other sources answer 403 Forbidden — verified against
en.wikipedia.org. Sending a descriptive User-Agent (as Wikimedia's bot
policy asks for) is the whole fix; everything else is inherited."""
USER_AGENT = (
"GAIA-Agent/1.0 (HF Agents Course Unit 4 final assignment; "
"+https://huggingface.co/spaces/sumit1703/Final_Assignment_Sumit)"
)
def forward(self, url: str) -> str:
import re
import requests as _requests
from markdownify import markdownify
from requests.exceptions import RequestException
try:
response = _requests.get(
url, timeout=20, headers={"User-Agent": self.USER_AGENT}
)
response.raise_for_status()
markdown_content = markdownify(response.text).strip()
markdown_content = re.sub(r"\n{3,}", "\n\n", markdown_content)
return self._truncate_content(markdown_content, self.max_output_length)
except _requests.exceptions.Timeout:
return "The request timed out. Please try again later or check the URL."
except RequestException as e:
return f"Error fetching the webpage: {str(e)}"
except Exception as e:
return f"An unexpected error occurred: {str(e)}"
class GAIAAgent:
"""
Wraps a smolagents CodeAgent for GAIA-style questions: general reasoning,
Python for calculation/data processing, web search, and local-file
inspection when a question ships an attachment.
"""
def __init__(self):
hf_token = os.environ.get("HF_TOKEN")
model_id = os.environ.get("GAIA_MODEL_ID") # optional override
provider = os.environ.get("GAIA_PROVIDER") # optional override; unset = library default ("auto")
base_url = os.environ.get("GAIA_BASE_URL") # optional OpenAI-compatible endpoint
api_key = os.environ.get("GAIA_API_KEY") # key for that endpoint
rpm = float(os.environ.get("GAIA_RPM") or DEFAULT_REQUESTS_PER_MINUTE)
# GAIA_MODEL_ID may list several models, best first. Later entries are
# used only when an earlier one exhausts its daily token quota.
model_ids = [m.strip() for m in (model_id or "").split(",") if m.strip()]
model_id = model_ids[0] if model_ids else None
fallback_model_ids = model_ids[1:]
# Two ways to reach a model, both through the same InferenceClientModel:
# 1. (default) HF Inference Providers, billed to HF_TOKEN's account.
# 2. any OpenAI-compatible endpoint, when GAIA_BASE_URL is set. This
# exists because HF's free monthly credits are small and a
# depleted account returns 402 on every single call, which fails
# all 20 questions at once.
# Fail fast and loud either way, instead of letting all 20 questions die
# silently at Step 1 with a generic error from deep inside smolagents.
if base_url:
print(f"Using custom OpenAI-compatible endpoint: {base_url}")
print(f"GAIA_API_KEY present: {bool(api_key)}")
if not api_key:
raise RuntimeError(
"GAIA_BASE_URL is set but GAIA_API_KEY is not. Add a secret "
"named exactly GAIA_API_KEY holding the key for that endpoint."
)
if not model_id:
raise RuntimeError(
"GAIA_BASE_URL is set but GAIA_MODEL_ID is not. A custom "
"endpoint needs its own model name (for example "
"'llama-3.3-70b-versatile' on Groq), since provider model "
"ids differ from Hugging Face repo ids."
)
# Bound each HTTP call: the openai SDK otherwise waits up to 10
# minutes and silently retries, so one throttled call can stall a
# whole question. Our own retry loop handles the backoff instead.
model = RetryingOpenAIServerModel(
model_id=model_id,
api_base=base_url,
api_key=api_key,
requests_per_minute=rpm,
retry=False,
client_kwargs={"timeout": 90, "max_retries": 0},
)
model_kwargs = {"model_id": model_id}
else:
print(f"HF_TOKEN present: {bool(hf_token)}")
if not hf_token:
raise RuntimeError(
"HF_TOKEN is not set in this process. Go to Space Settings > "
"Variables and secrets and confirm a secret named exactly "
"HF_TOKEN exists, then restart the Space (adding a secret does "
"not always hot-reload a running container)."
)
model_kwargs = {"token": hf_token, "model_id": model_id or DEFAULT_MODEL_ID}
if provider:
model_kwargs["provider"] = provider
model = RetryingInferenceClientModel(
requests_per_minute=rpm, retry=False, **model_kwargs
)
# Instance attribute, so exhausting one model's quota never mutates the
# class-level default shared by every other agent in the process.
model.fallback_model_ids = list(fallback_model_ids)
print(f"Model: {model_kwargs['model_id']} (paced at {rpm:g} requests/minute)")
if fallback_model_ids:
print(f"Fallback models on daily quota exhaustion: {', '.join(fallback_model_ids)}")
self.agent = CodeAgent(
tools=[],
model=model,
add_base_tools=True, # gives web_search + visit_webpage
additional_authorized_imports=ADDITIONAL_IMPORTS,
max_steps=MAX_STEPS,
step_callbacks=[trim_observations],
)
# add_base_tools installs the stock tools last, so swap in the hardened
# subclasses afterwards rather than passing them via tools=[].
# Both are given smaller output budgets than the library defaults
# (10 results, 40000 characters). Every tool result is replayed into the
# next model call, so a single stock visit_webpage is roughly 10k tokens
# — most of a free tier's whole per-minute allowance, spent on page
# boilerplate. Trimming keeps trajectories inside the budget and makes
# each step cheaper without losing the part of the page that matters.
self.agent.tools["web_search"] = RetryingWebSearchTool(max_results=6)
self.agent.tools["visit_webpage"] = IdentifiedVisitWebpageTool(
max_output_length=20000
)
print("GAIAAgent initialized (smolagents CodeAgent).")
def __call__(
self,
question: str,
file_path: Optional[str] = None,
file_name: Optional[str] = None,
) -> str:
task = question + "\n\n" + TASK_RULES
if file_path:
task += (
f"\nAn attachment for this task was downloaded to this local "
f"path: {file_path}\nOpen it with Python and inspect its "
f"contents before answering.\n"
)
elif file_name:
# The question ships an attachment but the scoring API could not
# serve it. Say so, otherwise the agent invents file contents.
task += (
f"\nNOTE: this task references an attachment ({file_name}) but "
f"it could not be retrieved from the evaluation server, so you "
f"do not have it. Do not pretend to open or read it. Answer "
f"from the question text and web research alone if that is "
f"possible; otherwise give your best supported answer.\n"
)
raw_answer = self.agent.run(task)
return clean_final_answer(raw_answer)
def clean_final_answer(raw) -> str:
"""Light, conservative cleanup only — no aggressive normalization that
could alter a valid exact-match answer."""
text = str(raw).strip()
if len(text) >= 2 and text[0] == text[-1] and text[0] in "\"'":
text = text[1:-1].strip()
for prefix in ("FINAL ANSWER:", "Final answer:", "Answer:", "answer:"):
if text.startswith(prefix):
text = text[len(prefix):].strip()
return text
def download_task_file(api_url: str, task_id: str, file_name: str) -> Optional[str]:
"""Downloads the attachment for a task via the existing /files/{task_id}
endpoint. Returns a local path, or None if there's no file or the
download fails (failure here must not crash the whole run)."""
if not file_name:
return None
try:
resp = requests.get(f"{api_url}/files/{task_id}", timeout=30)
if resp.status_code == 404:
# Distinguish "the evaluation server has no file mapped for this
# task" from a transport failure — they need different follow-ups.
print(
f"Attachment unavailable for task {task_id} ({file_name}): "
f"server returned 404 ({resp.text[:200]})"
)
return None
resp.raise_for_status()
out_dir = Path(tempfile.gettempdir()) / "gaia_files"
out_dir.mkdir(exist_ok=True)
out_path = out_dir / file_name
out_path.write_bytes(resp.content)
print(f"Downloaded attachment for task {task_id}: {out_path} ({len(resp.content)} bytes)")
return str(out_path)
except Exception as e:
print(f"Could not download file for task {task_id}: {type(e).__name__}: {e}")
return None
def run_single_question(task_id: str = ""):
"""Development helper: pull one task (a specific task_id, or a random one
from /random-question when left blank), run the agent on it, and show the
result. Submits nothing — this exists so individual tasks can be debugged
without spending a full 20-question submission."""
api_url = DEFAULT_API_URL
try:
task_id = (task_id or "").strip()
if task_id:
resp = requests.get(f"{api_url}/questions", timeout=15)
resp.raise_for_status()
matches = [q for q in resp.json() if q.get("task_id") == task_id]
if not matches:
return f"No question found with task_id {task_id}.", ""
item = matches[0]
else:
resp = requests.get(f"{api_url}/random-question", timeout=15)
resp.raise_for_status()
item = resp.json()
except Exception as e:
return f"Error fetching question: {type(e).__name__}: {e}", ""
task_id = item.get("task_id")
question_text = item.get("question")
file_name = item.get("file_name")
file_path = download_task_file(api_url, task_id, file_name) if file_name else None
if not file_name:
attachment_status = "none"
elif file_path:
attachment_status = f"{file_name} -> {file_path}"
else:
attachment_status = f"{file_name} (UNAVAILABLE - server returned no file)"
header = f"Task ID: {task_id}\nAttachment: {attachment_status}\n\nQuestion:\n{question_text}"
try:
agent = GAIAAgent()
except Exception as e:
return f"{header}\n\nError initializing agent: {e}", ""
try:
answer = agent(question_text, file_path=file_path, file_name=file_name)
return header, answer
except Exception as e:
print(f"Error running agent on task {task_id}: {type(e).__name__}: {e}")
return header, f"AGENT ERROR: {type(e).__name__}: {e}"
def run_and_submit_all(profile: gr.OAuthProfile | None):
"""
Fetches all questions, runs the GAIAAgent on them (downloading any
attachment first), submits all answers, and displays the results.
"""
space_id = os.getenv("SPACE_ID")
if profile:
username = f"{profile.username}"
print(f"User logged in: {username}")
else:
print("User not logged in.")
return "Please Login to Hugging Face with the button.", None
api_url = DEFAULT_API_URL
questions_url = f"{api_url}/questions"
submit_url = f"{api_url}/submit"
try:
agent = GAIAAgent()
except Exception as e:
print(f"Error instantiating agent: {e}")
return f"Error initializing agent: {e}", None
agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main"
print(agent_code)
print(f"Fetching questions from: {questions_url}")
try:
response = requests.get(questions_url, timeout=15)
response.raise_for_status()
questions_data = response.json()
if not questions_data:
print("Fetched questions list is empty.")
return "Fetched questions list is empty or invalid format.", None
print(f"Fetched {len(questions_data)} questions.")
except requests.exceptions.RequestException as e:
print(f"Error fetching questions: {e}")
return f"Error fetching questions: {e}", None
except requests.exceptions.JSONDecodeError as e:
print(f"Error decoding JSON response from questions endpoint: {e}")
print(f"Response text: {response.text[:500]}")
return f"Error decoding server response for questions: {e}", None
except Exception as e:
print(f"An unexpected error occurred fetching questions: {e}")
return f"An unexpected error occurred fetching questions: {e}", None
results_log = []
answers_payload = []
print(f"Running agent on {len(questions_data)} questions...")
for item in questions_data:
task_id = item.get("task_id")
question_text = item.get("question")
file_name = item.get("file_name")
if not task_id or question_text is None:
print(f"Skipping item with missing task_id or question: {item}")
continue
file_path = None
if file_name:
file_path = download_task_file(api_url, task_id, file_name)
try:
submitted_answer = agent(question_text, file_path=file_path, file_name=file_name)
answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer})
results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer})
except Exception as e:
print(f"Error running agent on task {task_id}: {type(e).__name__}: {e}")
results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"})
if not answers_payload:
print("Agent did not produce any answers to submit.")
return "Agent did not produce any answers to submit.", pd.DataFrame(results_log)
submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload}
status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..."
print(status_update)
print(f"Submitting {len(answers_payload)} answers to: {submit_url}")
try:
response = requests.post(submit_url, json=submission_data, timeout=60)
response.raise_for_status()
result_data = response.json()
final_status = (
f"Submission Successful!\n"
f"User: {result_data.get('username')}\n"
f"Overall Score: {result_data.get('score', 'N/A')}% "
f"({result_data.get('correct_count', '?')}/{result_data.get('total_attempted', '?')} correct)\n"
f"Message: {result_data.get('message', 'No message received.')}"
)
print("Submission successful.")
results_df = pd.DataFrame(results_log)
return final_status, results_df
except requests.exceptions.HTTPError as e:
error_detail = f"Server responded with status {e.response.status_code}."
try:
error_json = e.response.json()
error_detail += f" Detail: {error_json.get('detail', e.response.text)}"
except requests.exceptions.JSONDecodeError:
error_detail += f" Response: {e.response.text[:500]}"
status_message = f"Submission Failed: {error_detail}"
print(status_message)
results_df = pd.DataFrame(results_log)
return status_message, results_df
except requests.exceptions.Timeout:
status_message = "Submission Failed: The request timed out."
print(status_message)
results_df = pd.DataFrame(results_log)
return status_message, results_df
except requests.exceptions.RequestException as e:
status_message = f"Submission Failed: Network error - {e}"
print(status_message)
results_df = pd.DataFrame(results_log)
return status_message, results_df
except Exception as e:
status_message = f"An unexpected error occurred during submission: {e}"
print(status_message)
results_df = pd.DataFrame(results_log)
return status_message, results_df
# --- Build Gradio Interface using Blocks ---
with gr.Blocks() as demo:
gr.Markdown("# Basic Agent Evaluation Runner")
gr.Markdown(
"""
**Instructions:**
1. Log in to your Hugging Face account using the button below.
2. Click 'Run Evaluation & Submit All Answers' to fetch questions, run the agent,
submit answers, and see the score.
---
**Disclaimers:**
This can take a while — the agent works through each GAIA question in turn,
calling the model and, when needed, web search or Python execution.
"""
)
gr.LoginButton()
run_button = gr.Button("Run Evaluation & Submit All Answers")
status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False)
results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True)
run_button.click(
fn=run_and_submit_all,
outputs=[status_output, results_table]
)
with gr.Accordion("Developer: test a single question (no submission)", open=False):
gr.Markdown(
"Leave the box empty to pull a random task from `/random-question`, "
"or paste a specific `task_id` from `/questions`."
)
single_task_id = gr.Textbox(label="task_id (optional)", placeholder="leave blank for a random question")
single_button = gr.Button("Test One Question")
single_question_output = gr.Textbox(label="Task / Question", lines=8, interactive=False)
single_answer_output = gr.Textbox(label="Agent Answer (cleaned)", lines=3, interactive=False)
single_button.click(
fn=run_single_question,
inputs=[single_task_id],
outputs=[single_question_output, single_answer_output],
)
if __name__ == "__main__":
print("\n" + "-" * 30 + " App Starting " + "-" * 30)
space_host_startup = os.getenv("SPACE_HOST")
space_id_startup = os.getenv("SPACE_ID")
if space_host_startup:
print(f"✅ SPACE_HOST found: {space_host_startup}")
print(f" Runtime URL: https://{space_host_startup}")
else:
print("ℹ️ SPACE_HOST environment variable not found (running locally?).")
if space_id_startup:
print(f"✅ SPACE_ID found: {space_id_startup}")
print(f" Repo URL: https://huggingface.co/spaces/{space_id_startup}")
print(f" Repo Tree URL: https://huggingface.co/spaces/{space_id_startup}/tree/main")
else:
print("ℹ️ SPACE_ID environment variable not found (running locally?). Repo URL cannot be determined.")
print("-" * (60 + len(" App Starting ")) + "\n")
print("Launching Gradio Interface for Basic Agent Evaluation...")
demo.launch(debug=True, share=False) |