File size: 30,856 Bytes
06094f8
 
10e9b7d
ef2c544
06094f8
ef2c544
06094f8
 
 
10e9b7d
eccf8e4
3c4371f
10e9b7d
ef2c544
 
06094f8
e80aab9
3db6293
e80aab9
ef2c544
 
 
 
 
06094f8
 
 
 
 
 
 
 
ef2c544
 
 
 
 
 
06094f8
 
 
 
 
 
 
 
 
 
 
 
ef2c544
 
 
 
 
 
 
 
06094f8
ef2c544
06094f8
 
ef2c544
 
 
06094f8
 
 
ef2c544
06094f8
 
 
ef2c544
06094f8
 
ef2c544
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
06094f8
 
 
 
 
 
 
 
31243f4
06094f8
 
991cedd
ef2c544
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
c489c3a
 
ef2c544
 
 
 
 
 
 
06094f8
 
 
 
ef2c544
06094f8
 
ef2c544
 
 
 
 
 
 
 
 
 
 
 
 
06094f8
 
 
ef2c544
 
 
 
 
 
06094f8
 
 
 
 
 
 
ef2c544
 
 
 
 
 
 
 
 
 
06094f8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ef2c544
 
 
 
 
 
 
 
06094f8
 
 
 
 
ef2c544
06094f8
 
ef2c544
06094f8
 
 
ef2c544
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
06094f8
31243f4
06094f8
 
31243f4
06094f8
3c4371f
7e4a06b
06094f8
3c4371f
7e4a06b
3c4371f
7d65c66
3c4371f
7e4a06b
31243f4
 
e80aab9
31243f4
06094f8
31243f4
3c4371f
31243f4
06094f8
36ed51a
c1fd3d2
3c4371f
31243f4
eccf8e4
31243f4
7d65c66
31243f4
 
06094f8
 
31243f4
e80aab9
31243f4
 
3c4371f
06094f8
 
 
7d65c66
31243f4
 
e80aab9
7d65c66
 
3c4371f
31243f4
 
 
06094f8
31243f4
 
 
06094f8
 
 
 
 
31243f4
ef2c544
7d65c66
 
31243f4
ef2c544
06094f8
31243f4
 
3c4371f
31243f4
 
7d65c66
3c4371f
31243f4
e80aab9
31243f4
e80aab9
7d65c66
e80aab9
 
31243f4
e80aab9
 
3c4371f
 
 
e80aab9
 
31243f4
 
e80aab9
3c4371f
e80aab9
 
3c4371f
e80aab9
7d65c66
3c4371f
31243f4
7d65c66
31243f4
3c4371f
 
 
 
 
e80aab9
31243f4
 
 
 
7d65c66
31243f4
 
 
 
e80aab9
 
 
 
31243f4
0ee0419
e514fd7
 
06094f8
 
 
e514fd7
 
 
06094f8
 
e514fd7
e80aab9
 
7e4a06b
e80aab9
31243f4
e80aab9
9088b99
7d65c66
e80aab9
31243f4
 
 
e80aab9
 
ef2c544
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e80aab9
06094f8
3c4371f
06094f8
7d65c66
3c4371f
 
ef2c544
3c4371f
06094f8
7d65c66
06094f8
7d65c66
 
 
 
06094f8
7d65c66
06094f8
3c4371f
31243f4
3c4371f
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
import spaces  # must be the first import — required by ZeroGPU Spaces

import os
import re
import tempfile
import time
from pathlib import Path
from typing import Optional

import gradio as gr
import requests
import pandas as pd

from smolagents import CodeAgent, InferenceClientModel, OpenAIServerModel
from smolagents.default_tools import DuckDuckGoSearchTool, VisitWebpageTool

# --- Constants ---
DEFAULT_API_URL = "https://agents-course-unit4-scoring.hf.space"

# Pinned explicitly rather than relying on InferenceClientModel's own default,
# which changes between smolagents releases. The GAIA_MODEL_ID env var still
# overrides this if a different model is ever needed.
DEFAULT_MODEL_ID = "meta-llama/Llama-3.3-70B-Instruct"

# This Space's free-tier hardware is ZeroGPU, which refuses to start unless
# at least one function is decorated with @spaces.GPU. The agent itself never
# touches a GPU (it only calls HF Inference Providers over HTTP), so this
# function exists purely to satisfy that startup check and is never called.
@spaces.GPU
def _zerogpu_startup_check():
    return None

# Free provider tiers cap requests per minute (Google's Gemini free tier
# reports "limit: 20" for gemini-2.5-flash). Pacing our own calls below that
# ceiling is far cheaper than discovering it via 429s and backoff. Override
# with GAIA_RPM when moving to a tier with a different budget.
DEFAULT_REQUESTS_PER_MINUTE = 15.0

# Instructions appended to every question so the CodeAgent behaves like a
# GAIA-solving agent rather than a free-form chatbot. Kept as plain task
# framing (not a tool) to avoid adding moving parts.
TASK_RULES = """
Rules:
1. Think step by step. Break multi-part questions into sub-steps before acting.
2. Use web search for anything requiring current facts, names, dates, or
   information you are not already certain of. Never invent facts or sources.
3. Use Python (you write code as your actions) for arithmetic, counting,
   sorting, filtering, or any data processing rather than doing it in your head.
4. If an attachment path is given below, open and inspect it with Python
   before answering — never guess its contents.
5. If web_search fails or comes back empty, do not give up on the step: retry
   with a differently worded query, or call visit_webpage directly on a likely
   source URL (for example
   https://en.wikipedia.org/w/index.php?search=YOUR+QUERY, or the article URL
   itself). Do not try to import external packages such as wikipedia,
   requests-html, bs4 or googlesearch — they are not installed, and attempting
   them only wastes a step.
6. Where practical, verify an important intermediate result a second way
   before finalizing.
7. Call final_answer(...) with ONLY the exact value requested:
   - a bare number (no commas, no units unless the question asks for units)
   - a bare string (no "The answer is", no trailing period, no quotes)
   - a comma-separated list only if asked for a list, in exactly the order the
     question asks for; separate elements with a comma and a single space
     unless the question specifies a different separator
   Do not include your reasoning or the words "FINAL ANSWER" in that value.
"""

MAX_STEPS = 12
ADDITIONAL_IMPORTS = [
    "pandas", "numpy", "csv", "json", "re", "math", "statistics",
    "datetime", "itertools", "collections", "io", "os", "pypdf", "openpyxl",
    "requests", "unicodedata", "time",
]

# Every past step's observations are replayed into every later model call, so a
# trajectory grows quadratically and eventually exceeds what a free tier will
# accept in one request — Groq returns HTTP 413 "Request too large ... Limit
# 12000, Requested 12396" and no amount of waiting fixes it, because a single
# call already exceeds the whole per-minute budget. Recent observations are what
# the agent is reasoning about; older ones only need to stay recognizable.
RECENT_STEPS_KEPT_FULL = 2
RECENT_OBSERVATION_CHARS = 8000
OLDER_OBSERVATION_CHARS = 1500
# The model's own output is replayed as an assistant message too, and reasoning
# models are verbose enough that untrimmed history alone can breach the limit.
# The code it wrote is the part worth keeping, so older steps keep only a head.
RECENT_OUTPUT_CHARS = 4000
OLDER_OUTPUT_CHARS = 1000


def trim_observations(memory_step, agent=None) -> None:
    """Step callback: shrink observations of all but the newest few steps.

    Registered on ActionStep, so it runs after each action and keeps the next
    request inside the provider's per-request ceiling.
    """
    def _clip(text, budget):
        if not isinstance(text, str) or len(text) <= budget:
            return text
        return (
            text[:budget]
            + f"\n...[{len(text) - budget} characters truncated to stay within "
              f"the model's request size limit]"
        )

    steps = [s for s in getattr(agent, "memory", None).steps if hasattr(s, "observations")]
    for index, step in enumerate(steps):
        is_recent = index >= len(steps) - RECENT_STEPS_KEPT_FULL
        if step.observations:
            step.observations = _clip(
                step.observations,
                RECENT_OBSERVATION_CHARS if is_recent else OLDER_OBSERVATION_CHARS,
            )
        if step.model_output:
            step.model_output = _clip(
                step.model_output,
                RECENT_OUTPUT_CHARS if is_recent else OLDER_OUTPUT_CHARS,
            )


class RetryingWebSearchTool(DuckDuckGoSearchTool):
    """Same tool the base toolkit provides, but a rate-limited or flaky
    DuckDuckGo response is retried instead of burning one of the agent's
    steps. Name/description/signature are inherited unchanged, so the agent's
    system prompt is identical to the stock tool's."""

    max_attempts = 3
    backoff_seconds = 3.0

    def forward(self, query: str) -> str:
        last_error = None
        for attempt in range(self.max_attempts):
            try:
                return super().forward(query)
            except Exception as e:
                last_error = e
                print(f"web_search attempt {attempt + 1}/{self.max_attempts} failed: {e}")
                if attempt < self.max_attempts - 1:
                    time.sleep(self.backoff_seconds * (attempt + 1))
        raise RuntimeError(
            f"web_search failed after {self.max_attempts} attempts: {last_error}. "
            "Try visit_webpage on a likely source URL instead."
        )


class _RetryOnThrottleMixin:
    """Retries transient rate-limit / server errors from the inference endpoint.
    Free provider tiers throttle aggressively, and without this a single 429
    aborts the whole question. Non-transient errors (401, 402 out of credits,
    400 bad request) are re-raised immediately — retrying those is pointless.

    Both model classes below are constructed with retry=False, which disables
    smolagents' own retryer. Leaving it on nests two exponential backoffs: its
    3 attempts (60s base, doubling, jittered) run inside each of our attempts,
    so one throttled call can sleep for over half an hour. This layer replaces
    it because it can read the provider's own retry hint instead of guessing."""

    max_attempts = 6
    backoff_seconds = 20.0
    RETRYABLE = (
        "429", "500", "502", "503", "504", "rate limit", "too many requests", "timeout",
        # Some models occasionally emit a tool call where CodeAgent expects a
        # code block, and the provider rejects the request outright with
        # 400 "Tool choice is none, but model called a tool". It is a sampling
        # artifact rather than a bad prompt, so re-asking usually succeeds.
        "tool_use_failed",
    )
    # A per-day quota and an over-sized single request both arrive dressed as
    # rate limits, but neither clears within any backoff we would sit through:
    # the daily bucket refills hours later, and a request that alone exceeds the
    # per-minute ceiling will be exactly as large on the next attempt. Failing
    # this question immediately leaves budget and wall-clock for the rest.
    FATAL = ("tokens per day", "tpd", "request too large", "413")
    # Providers usually say how long to wait; obeying that beats guessing.
    # Gemini phrases it "Please retry in 32.290648364s", OpenAI-style APIs
    # "retry after 12 seconds".
    _RETRY_HINT = re.compile(r"retry(?:\s+after)?\s+in\s+([0-9.]+)\s*s|retry after ([0-9.]+)")

    # Filled in by GAIAAgent when GAIA_MODEL_ID lists more than one model.
    fallback_model_ids: list = []

    def generate(self, *args, **kwargs):
        last_error = None
        for attempt in range(self.max_attempts):
            try:
                return super().generate(*args, **kwargs)
            except Exception as e:
                message = str(e).lower()
                if any(token in message for token in self.FATAL):
                    # Daily quotas are per-model, so a sibling model on the same
                    # account usually still has budget. Switching costs one call
                    # and rescues every remaining question; without it the run
                    # ends here no matter how much wall-clock is left.
                    if "tokens per day" in message and self.fallback_model_ids:
                        self.model_id = self.fallback_model_ids.pop(0)
                        print(
                            f"Daily token quota exhausted; switching to "
                            f"fallback model {self.model_id}"
                        )
                        continue
                    raise
                if not any(token in message for token in self.RETRYABLE):
                    raise
                last_error = e
                # A malformed generation clears on the next sample, so re-ask
                # straight away rather than serving a rate-limit-sized backoff.
                if "tool_use_failed" in message:
                    wait = 1.0
                else:
                    wait = self.backoff_seconds * (attempt + 1)
                hint = self._RETRY_HINT.search(message)
                if hint:
                    # +2s of slack so we come back after the window, not on its edge
                    wait = max(wait, float(hint.group(1) or hint.group(2)) + 2.0)
                print(
                    f"Model call attempt {attempt + 1}/{self.max_attempts} failed "
                    f"({type(e).__name__}: {str(e)[:200]}); retrying in {wait:.0f}s"
                )
                if attempt < self.max_attempts - 1:
                    time.sleep(wait)
        raise last_error


class RetryingInferenceClientModel(_RetryOnThrottleMixin, InferenceClientModel):
    """HF Inference Providers, with throttle retries."""


class RetryingOpenAIServerModel(_RetryOnThrottleMixin, OpenAIServerModel):
    """Any OpenAI-compatible endpoint, with throttle retries. InferenceClient
    itself cannot take a model name and a base_url together, so custom
    endpoints go through this class instead."""


class IdentifiedVisitWebpageTool(VisitWebpageTool):
    """The stock tool calls requests.get() with no User-Agent, so Wikipedia and
    several other sources answer 403 Forbidden — verified against
    en.wikipedia.org. Sending a descriptive User-Agent (as Wikimedia's bot
    policy asks for) is the whole fix; everything else is inherited."""

    USER_AGENT = (
        "GAIA-Agent/1.0 (HF Agents Course Unit 4 final assignment; "
        "+https://huggingface.co/spaces/sumit1703/Final_Assignment_Sumit)"
    )

    def forward(self, url: str) -> str:
        import re

        import requests as _requests
        from markdownify import markdownify
        from requests.exceptions import RequestException

        try:
            response = _requests.get(
                url, timeout=20, headers={"User-Agent": self.USER_AGENT}
            )
            response.raise_for_status()
            markdown_content = markdownify(response.text).strip()
            markdown_content = re.sub(r"\n{3,}", "\n\n", markdown_content)
            return self._truncate_content(markdown_content, self.max_output_length)
        except _requests.exceptions.Timeout:
            return "The request timed out. Please try again later or check the URL."
        except RequestException as e:
            return f"Error fetching the webpage: {str(e)}"
        except Exception as e:
            return f"An unexpected error occurred: {str(e)}"


class GAIAAgent:
    """
    Wraps a smolagents CodeAgent for GAIA-style questions: general reasoning,
    Python for calculation/data processing, web search, and local-file
    inspection when a question ships an attachment.
    """

    def __init__(self):
        hf_token = os.environ.get("HF_TOKEN")
        model_id = os.environ.get("GAIA_MODEL_ID")  # optional override
        provider = os.environ.get("GAIA_PROVIDER")  # optional override; unset = library default ("auto")
        base_url = os.environ.get("GAIA_BASE_URL")  # optional OpenAI-compatible endpoint
        api_key = os.environ.get("GAIA_API_KEY")    # key for that endpoint
        rpm = float(os.environ.get("GAIA_RPM") or DEFAULT_REQUESTS_PER_MINUTE)

        # GAIA_MODEL_ID may list several models, best first. Later entries are
        # used only when an earlier one exhausts its daily token quota.
        model_ids = [m.strip() for m in (model_id or "").split(",") if m.strip()]
        model_id = model_ids[0] if model_ids else None
        fallback_model_ids = model_ids[1:]

        # Two ways to reach a model, both through the same InferenceClientModel:
        #   1. (default) HF Inference Providers, billed to HF_TOKEN's account.
        #   2. any OpenAI-compatible endpoint, when GAIA_BASE_URL is set. This
        #      exists because HF's free monthly credits are small and a
        #      depleted account returns 402 on every single call, which fails
        #      all 20 questions at once.
        # Fail fast and loud either way, instead of letting all 20 questions die
        # silently at Step 1 with a generic error from deep inside smolagents.
        if base_url:
            print(f"Using custom OpenAI-compatible endpoint: {base_url}")
            print(f"GAIA_API_KEY present: {bool(api_key)}")
            if not api_key:
                raise RuntimeError(
                    "GAIA_BASE_URL is set but GAIA_API_KEY is not. Add a secret "
                    "named exactly GAIA_API_KEY holding the key for that endpoint."
                )
            if not model_id:
                raise RuntimeError(
                    "GAIA_BASE_URL is set but GAIA_MODEL_ID is not. A custom "
                    "endpoint needs its own model name (for example "
                    "'llama-3.3-70b-versatile' on Groq), since provider model "
                    "ids differ from Hugging Face repo ids."
                )
            # Bound each HTTP call: the openai SDK otherwise waits up to 10
            # minutes and silently retries, so one throttled call can stall a
            # whole question. Our own retry loop handles the backoff instead.
            model = RetryingOpenAIServerModel(
                model_id=model_id,
                api_base=base_url,
                api_key=api_key,
                requests_per_minute=rpm,
                retry=False,
                client_kwargs={"timeout": 90, "max_retries": 0},
            )
            model_kwargs = {"model_id": model_id}
        else:
            print(f"HF_TOKEN present: {bool(hf_token)}")
            if not hf_token:
                raise RuntimeError(
                    "HF_TOKEN is not set in this process. Go to Space Settings > "
                    "Variables and secrets and confirm a secret named exactly "
                    "HF_TOKEN exists, then restart the Space (adding a secret does "
                    "not always hot-reload a running container)."
                )
            model_kwargs = {"token": hf_token, "model_id": model_id or DEFAULT_MODEL_ID}
            if provider:
                model_kwargs["provider"] = provider
            model = RetryingInferenceClientModel(
                requests_per_minute=rpm, retry=False, **model_kwargs
            )

        # Instance attribute, so exhausting one model's quota never mutates the
        # class-level default shared by every other agent in the process.
        model.fallback_model_ids = list(fallback_model_ids)

        print(f"Model: {model_kwargs['model_id']} (paced at {rpm:g} requests/minute)")
        if fallback_model_ids:
            print(f"Fallback models on daily quota exhaustion: {', '.join(fallback_model_ids)}")

        self.agent = CodeAgent(
            tools=[],
            model=model,
            add_base_tools=True,  # gives web_search + visit_webpage
            additional_authorized_imports=ADDITIONAL_IMPORTS,
            max_steps=MAX_STEPS,
            step_callbacks=[trim_observations],
        )
        # add_base_tools installs the stock tools last, so swap in the hardened
        # subclasses afterwards rather than passing them via tools=[].
        # Both are given smaller output budgets than the library defaults
        # (10 results, 40000 characters). Every tool result is replayed into the
        # next model call, so a single stock visit_webpage is roughly 10k tokens
        # — most of a free tier's whole per-minute allowance, spent on page
        # boilerplate. Trimming keeps trajectories inside the budget and makes
        # each step cheaper without losing the part of the page that matters.
        self.agent.tools["web_search"] = RetryingWebSearchTool(max_results=6)
        self.agent.tools["visit_webpage"] = IdentifiedVisitWebpageTool(
            max_output_length=20000
        )
        print("GAIAAgent initialized (smolagents CodeAgent).")

    def __call__(
        self,
        question: str,
        file_path: Optional[str] = None,
        file_name: Optional[str] = None,
    ) -> str:
        task = question + "\n\n" + TASK_RULES
        if file_path:
            task += (
                f"\nAn attachment for this task was downloaded to this local "
                f"path: {file_path}\nOpen it with Python and inspect its "
                f"contents before answering.\n"
            )
        elif file_name:
            # The question ships an attachment but the scoring API could not
            # serve it. Say so, otherwise the agent invents file contents.
            task += (
                f"\nNOTE: this task references an attachment ({file_name}) but "
                f"it could not be retrieved from the evaluation server, so you "
                f"do not have it. Do not pretend to open or read it. Answer "
                f"from the question text and web research alone if that is "
                f"possible; otherwise give your best supported answer.\n"
            )
        raw_answer = self.agent.run(task)
        return clean_final_answer(raw_answer)


def clean_final_answer(raw) -> str:
    """Light, conservative cleanup only — no aggressive normalization that
    could alter a valid exact-match answer."""
    text = str(raw).strip()
    if len(text) >= 2 and text[0] == text[-1] and text[0] in "\"'":
        text = text[1:-1].strip()
    for prefix in ("FINAL ANSWER:", "Final answer:", "Answer:", "answer:"):
        if text.startswith(prefix):
            text = text[len(prefix):].strip()
    return text


def download_task_file(api_url: str, task_id: str, file_name: str) -> Optional[str]:
    """Downloads the attachment for a task via the existing /files/{task_id}
    endpoint. Returns a local path, or None if there's no file or the
    download fails (failure here must not crash the whole run)."""
    if not file_name:
        return None
    try:
        resp = requests.get(f"{api_url}/files/{task_id}", timeout=30)
        if resp.status_code == 404:
            # Distinguish "the evaluation server has no file mapped for this
            # task" from a transport failure — they need different follow-ups.
            print(
                f"Attachment unavailable for task {task_id} ({file_name}): "
                f"server returned 404 ({resp.text[:200]})"
            )
            return None
        resp.raise_for_status()
        out_dir = Path(tempfile.gettempdir()) / "gaia_files"
        out_dir.mkdir(exist_ok=True)
        out_path = out_dir / file_name
        out_path.write_bytes(resp.content)
        print(f"Downloaded attachment for task {task_id}: {out_path} ({len(resp.content)} bytes)")
        return str(out_path)
    except Exception as e:
        print(f"Could not download file for task {task_id}: {type(e).__name__}: {e}")
        return None


def run_single_question(task_id: str = ""):
    """Development helper: pull one task (a specific task_id, or a random one
    from /random-question when left blank), run the agent on it, and show the
    result. Submits nothing — this exists so individual tasks can be debugged
    without spending a full 20-question submission."""
    api_url = DEFAULT_API_URL
    try:
        task_id = (task_id or "").strip()
        if task_id:
            resp = requests.get(f"{api_url}/questions", timeout=15)
            resp.raise_for_status()
            matches = [q for q in resp.json() if q.get("task_id") == task_id]
            if not matches:
                return f"No question found with task_id {task_id}.", ""
            item = matches[0]
        else:
            resp = requests.get(f"{api_url}/random-question", timeout=15)
            resp.raise_for_status()
            item = resp.json()
    except Exception as e:
        return f"Error fetching question: {type(e).__name__}: {e}", ""

    task_id = item.get("task_id")
    question_text = item.get("question")
    file_name = item.get("file_name")

    file_path = download_task_file(api_url, task_id, file_name) if file_name else None
    if not file_name:
        attachment_status = "none"
    elif file_path:
        attachment_status = f"{file_name} -> {file_path}"
    else:
        attachment_status = f"{file_name} (UNAVAILABLE - server returned no file)"

    header = f"Task ID: {task_id}\nAttachment: {attachment_status}\n\nQuestion:\n{question_text}"

    try:
        agent = GAIAAgent()
    except Exception as e:
        return f"{header}\n\nError initializing agent: {e}", ""

    try:
        answer = agent(question_text, file_path=file_path, file_name=file_name)
        return header, answer
    except Exception as e:
        print(f"Error running agent on task {task_id}: {type(e).__name__}: {e}")
        return header, f"AGENT ERROR: {type(e).__name__}: {e}"


def run_and_submit_all(profile: gr.OAuthProfile | None):
    """
    Fetches all questions, runs the GAIAAgent on them (downloading any
    attachment first), submits all answers, and displays the results.
    """
    space_id = os.getenv("SPACE_ID")

    if profile:
        username = f"{profile.username}"
        print(f"User logged in: {username}")
    else:
        print("User not logged in.")
        return "Please Login to Hugging Face with the button.", None

    api_url = DEFAULT_API_URL
    questions_url = f"{api_url}/questions"
    submit_url = f"{api_url}/submit"

    try:
        agent = GAIAAgent()
    except Exception as e:
        print(f"Error instantiating agent: {e}")
        return f"Error initializing agent: {e}", None

    agent_code = f"https://huggingface.co/spaces/{space_id}/tree/main"
    print(agent_code)

    print(f"Fetching questions from: {questions_url}")
    try:
        response = requests.get(questions_url, timeout=15)
        response.raise_for_status()
        questions_data = response.json()
        if not questions_data:
            print("Fetched questions list is empty.")
            return "Fetched questions list is empty or invalid format.", None
        print(f"Fetched {len(questions_data)} questions.")
    except requests.exceptions.RequestException as e:
        print(f"Error fetching questions: {e}")
        return f"Error fetching questions: {e}", None
    except requests.exceptions.JSONDecodeError as e:
        print(f"Error decoding JSON response from questions endpoint: {e}")
        print(f"Response text: {response.text[:500]}")
        return f"Error decoding server response for questions: {e}", None
    except Exception as e:
        print(f"An unexpected error occurred fetching questions: {e}")
        return f"An unexpected error occurred fetching questions: {e}", None

    results_log = []
    answers_payload = []
    print(f"Running agent on {len(questions_data)} questions...")
    for item in questions_data:
        task_id = item.get("task_id")
        question_text = item.get("question")
        file_name = item.get("file_name")
        if not task_id or question_text is None:
            print(f"Skipping item with missing task_id or question: {item}")
            continue

        file_path = None
        if file_name:
            file_path = download_task_file(api_url, task_id, file_name)

        try:
            submitted_answer = agent(question_text, file_path=file_path, file_name=file_name)
            answers_payload.append({"task_id": task_id, "submitted_answer": submitted_answer})
            results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": submitted_answer})
        except Exception as e:
            print(f"Error running agent on task {task_id}: {type(e).__name__}: {e}")
            results_log.append({"Task ID": task_id, "Question": question_text, "Submitted Answer": f"AGENT ERROR: {e}"})

    if not answers_payload:
        print("Agent did not produce any answers to submit.")
        return "Agent did not produce any answers to submit.", pd.DataFrame(results_log)

    submission_data = {"username": username.strip(), "agent_code": agent_code, "answers": answers_payload}
    status_update = f"Agent finished. Submitting {len(answers_payload)} answers for user '{username}'..."
    print(status_update)

    print(f"Submitting {len(answers_payload)} answers to: {submit_url}")
    try:
        response = requests.post(submit_url, json=submission_data, timeout=60)
        response.raise_for_status()
        result_data = response.json()
        final_status = (
            f"Submission Successful!\n"
            f"User: {result_data.get('username')}\n"
            f"Overall Score: {result_data.get('score', 'N/A')}% "
            f"({result_data.get('correct_count', '?')}/{result_data.get('total_attempted', '?')} correct)\n"
            f"Message: {result_data.get('message', 'No message received.')}"
        )
        print("Submission successful.")
        results_df = pd.DataFrame(results_log)
        return final_status, results_df
    except requests.exceptions.HTTPError as e:
        error_detail = f"Server responded with status {e.response.status_code}."
        try:
            error_json = e.response.json()
            error_detail += f" Detail: {error_json.get('detail', e.response.text)}"
        except requests.exceptions.JSONDecodeError:
            error_detail += f" Response: {e.response.text[:500]}"
        status_message = f"Submission Failed: {error_detail}"
        print(status_message)
        results_df = pd.DataFrame(results_log)
        return status_message, results_df
    except requests.exceptions.Timeout:
        status_message = "Submission Failed: The request timed out."
        print(status_message)
        results_df = pd.DataFrame(results_log)
        return status_message, results_df
    except requests.exceptions.RequestException as e:
        status_message = f"Submission Failed: Network error - {e}"
        print(status_message)
        results_df = pd.DataFrame(results_log)
        return status_message, results_df
    except Exception as e:
        status_message = f"An unexpected error occurred during submission: {e}"
        print(status_message)
        results_df = pd.DataFrame(results_log)
        return status_message, results_df


# --- Build Gradio Interface using Blocks ---
with gr.Blocks() as demo:
    gr.Markdown("# Basic Agent Evaluation Runner")
    gr.Markdown(
        """
        **Instructions:**
        1. Log in to your Hugging Face account using the button below.
        2. Click 'Run Evaluation & Submit All Answers' to fetch questions, run the agent,
           submit answers, and see the score.

        ---
        **Disclaimers:**
        This can take a while — the agent works through each GAIA question in turn,
        calling the model and, when needed, web search or Python execution.
        """
    )

    gr.LoginButton()

    run_button = gr.Button("Run Evaluation & Submit All Answers")

    status_output = gr.Textbox(label="Run Status / Submission Result", lines=5, interactive=False)
    results_table = gr.DataFrame(label="Questions and Agent Answers", wrap=True)

    run_button.click(
        fn=run_and_submit_all,
        outputs=[status_output, results_table]
    )

    with gr.Accordion("Developer: test a single question (no submission)", open=False):
        gr.Markdown(
            "Leave the box empty to pull a random task from `/random-question`, "
            "or paste a specific `task_id` from `/questions`."
        )
        single_task_id = gr.Textbox(label="task_id (optional)", placeholder="leave blank for a random question")
        single_button = gr.Button("Test One Question")
        single_question_output = gr.Textbox(label="Task / Question", lines=8, interactive=False)
        single_answer_output = gr.Textbox(label="Agent Answer (cleaned)", lines=3, interactive=False)

        single_button.click(
            fn=run_single_question,
            inputs=[single_task_id],
            outputs=[single_question_output, single_answer_output],
        )

if __name__ == "__main__":
    print("\n" + "-" * 30 + " App Starting " + "-" * 30)
    space_host_startup = os.getenv("SPACE_HOST")
    space_id_startup = os.getenv("SPACE_ID")

    if space_host_startup:
        print(f"✅ SPACE_HOST found: {space_host_startup}")
        print(f"   Runtime URL: https://{space_host_startup}")
    else:
        print("ℹ️ SPACE_HOST environment variable not found (running locally?).")

    if space_id_startup:
        print(f"✅ SPACE_ID found: {space_id_startup}")
        print(f"   Repo URL: https://huggingface.co/spaces/{space_id_startup}")
        print(f"   Repo Tree URL: https://huggingface.co/spaces/{space_id_startup}/tree/main")
    else:
        print("ℹ️ SPACE_ID environment variable not found (running locally?). Repo URL cannot be determined.")

    print("-" * (60 + len(" App Starting ")) + "\n")

    print("Launching Gradio Interface for Basic Agent Evaluation...")
    demo.launch(debug=True, share=False)