File size: 1,185 Bytes
ca20ec1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
import time


class UpstreamUnavailableError(Exception):
    """Raised when the LLM provider fails even after retries (e.g. persistent 503s)."""
    pass


def invoke_with_retry(chain, payload: dict, max_retries: int = 3, base_delay: float = 1.5):
    """Invokes a langchain chain with exponential backoff on transient failures.

    The Hugging Face router + featherless-ai provider can return a 503 "Service
    Unavailable" HTML page instead of JSON during cold starts or overload. The
    OpenAI-compatible client then throws while trying to parse that as a chat
    completion. Retrying with backoff handles the common transient case instead
    of failing the whole turn on the first hiccup.
    """
    last_error = None
    for attempt in range(1, max_retries + 1):
        try:
            return chain.invoke(payload)
        except Exception as e:
            last_error = e
            is_last = attempt == max_retries
            print(f"[llm_utils] attempt {attempt}/{max_retries} failed: {e}")
            if not is_last:
                time.sleep(base_delay * (2 ** (attempt - 1)))  # 1.5s, 3s, 6s...
    raise UpstreamUnavailableError(str(last_error)) from last_error