agentic_rag / config.py
Saint5's picture
Direct upload to ZeroGPU container
51e9502 verified
Raw
History Blame Contribute Delete
2.03 kB
"""
config.py
-------------
Central configuration for the Agentic RAG HuggingFace Space (ZeroGPU version).
Every other module imports from here. Keeping constants in one file
means changing a threshold or token limit is a one-line edit with
no risk of the value drifting out of sync across modules.
"""
import os
# --- Model identifiers -----------------------------------
# Back to 7B: ZeroGPU provides a real shared GPU, so quantized 7B
# inference is fast.
MODEL_ID = "Qwen/Qwen2.5-7B-Instruct"
EMBEDDING_MODEL_ID = "sentence-transformers/all-MiniLM-L6-v2"
# --- Inference settings ----------------------
MAX_NEW_TOKENS = 512 # Restored from 256 (CPU) — GPU has headroom for this
# --- ZeroGPU-specific setting -----------------
# Passed to @spaces.GPU(duration=ZEROGPU_DURATION) in generation.py.
# This is the max wall-clock time ZeroGPU allows a single decorated
# call before force-releasing the GPU slot back to the shared pool.
#
# Default in ZeroGPU is 60s if you don't specify duration. We set it
# explicitly to 120s to give a 7B model + 512 max_new_tokens headroom
# for a cold worker (first call after idle — weights stream from disk
# offload into VRAM, which adds latency beyond pure generation time).
#
# Tuning guidance: lower duration = higher queue priority for your
# Space's visitors (per ZeroGPU docs). Once you've benchmarked actual
# generation time in the Space logs, lower this toward that real number.
ZEROGPU_DURATION = 30 #
# --- Retrieval settings ----------------------
# FAISS returns L2 distances; we convert with sim = 1 / (1 + L2).
# 0.60 ≈ cosine similarity 0.78 — a solid topical match.
CONFIDENCE_THRESHOLD = 0.60
# Restored from 3000 (CPU) — GPU headroom allows the original budget.
MAX_PROMPT_TOKENS = 3500
# --- FAISS maintenance -----------------------------------
EXPIRY_DAYS = 1
# --- API keys --------------------------------------------
# Set these in Space Settings → Variables and secrets → New secret.
TAVILY_API_KEY = os.environ.get("TAVILY_API_KEY", "")