research-agent / agent /agents.py
pymite6941's picture
Exclude Venice provider via OpenRouter extra_body routing
8c3da34
Raw
History Blame Contribute Delete
4.46 kB
import os
from crewai import Agent, LLM
from agent.tools.search_tools import TavilySearchTool, ArxivSearchTool, SemanticScholarTool
from agent.tools.credibility_tool import CredibilityScorerTool
from agent.tools.rag_tool import EmbedSourcesTool, RetrieveChunksTool
import config
def _get_llm() -> LLM:
if config.OPENROUTER_API_KEY:
os.environ["OPENROUTER_API_KEY"] = config.OPENROUTER_API_KEY
return LLM(
model=config.OPENROUTER_MODEL,
temperature=0.1,
extra_body={"provider": {"ignore": ["Venice"]}},
)
if config.GROQ_API_KEY:
return LLM(
model=f"groq/{config.GROQ_MODEL}",
api_key=config.GROQ_API_KEY,
temperature=0.1,
)
raise RuntimeError("No LLM API key configured. Set OPENROUTER_API_KEY or GROQ_API_KEY.")
def build_agents() -> dict[str, Agent]:
llm = _get_llm()
researcher = Agent(
role="Senior Research Specialist",
goal=(
"Search multiple authoritative sources to gather comprehensive, "
"factual information on the research topic. Cover academic, government, "
"and high-quality journalistic sources. Gather raw_content whenever possible."
),
backstory=(
"You are a rigorous research specialist with a PhD-level background. "
"You know which sources to trust: peer-reviewed journals, government data, "
"established news wire services. You never rely on social media, opinion blogs, "
"or unverified personal sites. You systematically cover a topic from multiple angles."
),
tools=[TavilySearchTool(), ArxivSearchTool(), SemanticScholarTool()],
llm=llm,
verbose=True,
max_iter=8,
memory=False,
)
credibility_analyst = Agent(
role="Source Credibility Analyst",
goal=(
"Score every gathered source for credibility (0.0–1.0). "
"Drop sources below threshold. Flag low-confidence sources. "
"Return a clean, ranked list of accepted sources for synthesis."
),
backstory=(
"You are a fact-checker and media literacy expert. You evaluate sources "
"based on domain authority, HTTPS, publication date, author credentials, "
"and content quality. You are skeptical but fair — a government .edu source "
"gets high trust; a random Medium post without a byline gets low trust. "
"You always run the Credibility Scorer tool on the full source list."
),
tools=[CredibilityScorerTool()],
llm=llm,
verbose=True,
max_iter=3,
memory=False,
)
rag_engineer = Agent(
role="RAG Pipeline Engineer",
goal=(
"Embed accepted sources into the vector store, then retrieve the most "
"relevant chunks for the research question."
),
backstory=(
"You are a retrieval systems expert. You chunk documents optimally, "
"embed them, and retrieve precisely what the synthesis agent needs. "
"You always embed before retrieving. You use the session_id consistently."
),
tools=[EmbedSourcesTool(), RetrieveChunksTool()],
llm=llm,
verbose=True,
max_iter=4,
memory=False,
)
synthesis_analyst = Agent(
role="Research Synthesis Analyst",
goal=(
"Produce a structured, fully cited research report grounded only in retrieved chunks. "
"Every factual claim must have an inline [N] citation. "
"Never state facts not present in the provided chunks."
),
backstory=(
"You are a senior analyst who writes rigorous, cited research reports. "
"You have a zero-tolerance policy for hallucination: if the evidence does not "
"support a claim, you write 'INSUFFICIENT EVIDENCE' rather than guessing. "
"Your reports always end with a numbered reference list and a confidence label "
"(HIGH / MEDIUM / LOW / INSUFFICIENT)."
),
tools=[],
llm=llm,
verbose=True,
max_iter=3,
memory=False,
)
return {
"researcher": researcher,
"credibility_analyst": credibility_analyst,
"rag_engineer": rag_engineer,
"synthesis_analyst": synthesis_analyst,
}