Spaces:
Sleeping
Sleeping
File size: 5,247 Bytes
51abc3c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 | """High level multi-agent system powered by OpenRouter models.
This module sets up a manager agent that delegates tasks to specialized
web and information agents. It relies on the ``smolagent`` framework and
OpenRouter API models for language generation and verification.
"""
from smolagents import (
CodeAgent,
VisitWebpageTool,
WebSearchTool,
WikipediaSearchTool,
PythonInterpreterTool,
FinalAnswerTool,
OpenAIServerModel,
)
from smolagents.utils import encode_image_base64, make_image_url
from vision_tool import image_reasoning_tool
import os
OPENROUTER_API_KEY = os.getenv("OPENROUTER_API_KEY")
if not OPENROUTER_API_KEY:
raise EnvironmentError("OPENROUTER_API_KEY environment variable not set")
common = dict(
api_base="https://openrouter.ai/api/v1",
api_key=OPENROUTER_API_KEY,
)
class MultiAgentSystem:
"""Coordinates specialized agents and their underlying models.
The system instantiates a ``web_agent`` for browsing and data collection,
an ``info_agent`` for computation and image reasoning, and a
``manager_agent`` that plans tasks and verifies answers using several
OpenRouter models.
"""
def __init__(self):
self.deepseek_model = OpenAIServerModel(
model_id="deepseek/deepseek-r1-0528:free",
max_tokens=8096,
**common,
)
self.qwen_model = OpenAIServerModel(
model_id="qwen/qwen-2.5-coder-32b-instruct:free",
max_tokens=8096,
**common,
)
self.gemini_model = OpenAIServerModel(
model_id="google/gemini-2.0-flash-exp:free",
max_tokens=8096,
**common,
)
self.web_agent = CodeAgent(
model_id=self.qwen_model,
tools=[WebSearchTool(), VisitWebpageTool(), WikipediaSearchTool()],
name="web_agent",
description=(
"You are a web browsing agent. Whenever the given {task} involves browsing "
"the web or a specific website such as Wikipedia or YouTube, you will use "
"the provided tools. For web-based factual and retrieval tasks, be as precise and source-reliable as possible."
),
additional_authorized_imports=[
"markdownify",
"json",
"requests",
"urllib.request",
"urllib.parse",
"wikipedia-api",
],
verbosity_level=0,
max_steps=10,
)
self.info_agent = CodeAgent(
model_id=self.qwen_model,
tools=[PythonInterpreterTool(), image_reasoning_tool],
name="info_agent",
description=(
"You are an agent tasked with cleaning, parsing, calculating information, and performing OCR if images are provided in the {task}. "
"You can also analyze images using a vision model. You handle all math, code, and data manipulation. Use numpy, math, and available libraries. "
"For image or chess tasks, use pytesseract, PIL, chess, or the image_reasoning_tool as required."
),
additional_authorized_imports=[
"numpy",
"math",
"pytesseract",
"PIL",
"chess",
],
max_tokens=8096,
)
self.manager_agent = CodeAgent(
model_id=self.deepseek_model,
tools=[FinalAnswerTool()],
managed_agents=[self.web_agent, self.info_agent],
name="manager_agent",
description=(
"You are the manager. Given a {task}, plan which agent to use: "
"If web data is needed, delegate to web_agent. If math, parsing, image reasoning, or code is needed, use info_agent. "
"After collecting outputs, optionally cross-validate and check correctness, then finalize and submit the best answer using FinalAnswerTool. "
"For each task, explicitly explain your planning steps and reasons for choosing which agent, and always prefer the most accurate and complete answer possible."
),
additional_authorized_imports=[
"json",
"pandas",
"numpy",
],
planning_interval=3,
verbosity_level=2,
max_tokens=8096,
final_answer_check=[self.check_reasoning],
max_steps=8,
)
def check_reasoning(self, final_answer, agent_memory):
model_id = self.gemini_model
verification_prompt = (
f"Here is a user-given task and the agent steps: {agent_memory.get_succinct_steps()}. "
f"The proposed final answer is: {final_answer}. "
"Please check that the reasoning process is correct: do they correctly answer the given task? "
"First list reasons why yes/no, then write your final decision: PASS in caps lock if it is satisfactory, FAIL if it is not."
)
output = model(verification_prompt)
print("Feedback: ", output)
if "FAIL" in output:
raise Exception(output)
return True
|