Spaces:
Sleeping
Sleeping
File size: 1,185 Bytes
ca20ec1 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 | import time
class UpstreamUnavailableError(Exception):
"""Raised when the LLM provider fails even after retries (e.g. persistent 503s)."""
pass
def invoke_with_retry(chain, payload: dict, max_retries: int = 3, base_delay: float = 1.5):
"""Invokes a langchain chain with exponential backoff on transient failures.
The Hugging Face router + featherless-ai provider can return a 503 "Service
Unavailable" HTML page instead of JSON during cold starts or overload. The
OpenAI-compatible client then throws while trying to parse that as a chat
completion. Retrying with backoff handles the common transient case instead
of failing the whole turn on the first hiccup.
"""
last_error = None
for attempt in range(1, max_retries + 1):
try:
return chain.invoke(payload)
except Exception as e:
last_error = e
is_last = attempt == max_retries
print(f"[llm_utils] attempt {attempt}/{max_retries} failed: {e}")
if not is_last:
time.sleep(base_delay * (2 ** (attempt - 1))) # 1.5s, 3s, 6s...
raise UpstreamUnavailableError(str(last_error)) from last_error |