hello / app.py
Nipurn's picture
Update app.py
840e949 verified
Raw
History Blame Contribute Delete
29.4 kB
# -*- coding: utf-8 -*-
# @Time : 2025/1/1
# @Author : nipurnagarwal
# @Email : nipurnagarwal26@gmail.com
# @Project : browser-use-100XPrompt
# @FileName: 100XPrompt.py
import pdb
from dotenv import load_dotenv
load_dotenv()
import argparse
import os
import gradio as gr
import argparse
from gradio.themes import Base, Default, Soft, Monochrome, Glass
import asyncio
import os, glob
from browser_use.agent.service import Agent
from browser_use.browser.browser import Browser, BrowserConfig
from browser_use.browser.context import (
BrowserContextConfig,
BrowserContextWindowSize,
)
from playwright.async_api import async_playwright
from src.agent.custom_agent import CustomAgent
from src.agent.custom_prompts import CustomSystemPrompt
from src.browser.custom_browser import CustomBrowser
from src.browser.custom_context import BrowserContextConfig
from src.controller.custom_controller import CustomController
from src.utils import utils
from src.utils.utils import update_model_dropdown
from src.browser.config import BrowserPersistenceConfig
from src.browser.custom_browser import CustomBrowser
from src.browser.custom_context import CustomBrowserContext
from browser_use.browser.browser import BrowserConfig
from browser_use.browser.context import BrowserContextConfig, BrowserContextWindowSize
import platform
import logging
logger = logging.getLogger(__name__)
# Global variables for persistence and control
_global_browser = None
_global_browser_context = None
_current_agent = None
_stop_event = asyncio.Event()
_agent_state = None
class BrowserNotFoundError(Exception):
pass
class AgentState:
def __init__(self, task, step_info, browser_state=None):
self.task = task
self.step_info = step_info
self.browser_state = browser_state
self.last_result = None
self.message_history = None
async def stop_agent():
"""Stop the currently running agent"""
global _stop_event, _current_agent, _agent_state
if _current_agent:
_stop_event.set()
# Store current state
if hasattr(_current_agent, 'browser_context'):
browser_state = await _current_agent.browser_context.get_state(use_vision=_current_agent.use_vision)
_agent_state = AgentState(
task=_current_agent.task,
step_info=_current_agent.step_info if hasattr(_current_agent, 'step_info') else None,
browser_state=browser_state
)
# Store the last result and message history
if hasattr(_current_agent, '_last_result'):
_agent_state.last_result = _current_agent._last_result
if hasattr(_current_agent, 'message_manager'):
_agent_state.message_history = _current_agent.message_manager.get_messages()
return True
async def resume_agent(new_task):
"""Resume agent with new instructions"""
global _stop_event, _current_agent, _agent_state
if _agent_state and _current_agent:
_stop_event.clear()
# Update task while preserving state
_current_agent.task = new_task
if hasattr(_current_agent, 'step_info') and _current_agent.step_info:
_current_agent.step_info.task = new_task
# Preserve the step number and memory
_current_agent.step_info.step_number = _agent_state.step_info.step_number
_current_agent.step_info.memory = _agent_state.step_info.memory
_current_agent.step_info.task_progress = _agent_state.step_info.task_progress
# Restore browser state if available
if _agent_state.browser_state and hasattr(_current_agent, 'browser_context'):
# Update the message manager with the current state
_current_agent.message_manager.add_state_message(
_agent_state.browser_state,
_current_agent._last_result,
_current_agent.step_info
)
# Clear the stored state since we've restored it
_agent_state = None
return True
return False
async def run_browser_agent(
agent_type,
llm_provider,
llm_model_name,
llm_temperature,
llm_base_url,
llm_api_key,
use_own_browser,
browser_type,
keep_browser_open,
headless,
disable_security,
window_w,
window_h,
save_recording_path,
save_trace_path,
enable_recording,
task,
add_infos,
max_steps,
use_vision,
max_actions_per_step,
tool_call_in_content
):
global _current_agent, _stop_event
# Reset stop event
_stop_event = asyncio.Event()
# Disable recording if the checkbox is unchecked
if not enable_recording:
save_recording_path = None
# Ensure the recording directory exists if recording is enabled
if save_recording_path:
os.makedirs(save_recording_path, exist_ok=True)
# Get the list of existing videos before the agent runs
existing_videos = set()
if save_recording_path:
existing_videos = set(
glob.glob(os.path.join(save_recording_path, "*.[mM][pP]4"))
+ glob.glob(os.path.join(save_recording_path, "*.[wW][eE][bB][mM]"))
)
# Run the agent
llm = utils.get_llm_model(
provider=llm_provider,
model_name=llm_model_name,
temperature=llm_temperature,
base_url=llm_base_url,
api_key=llm_api_key,
)
try:
if agent_type == "org":
final_result, errors, model_actions, model_thoughts = await run_org_agent(
llm=llm,
use_own_browser=use_own_browser,
browser_type=browser_type,
keep_browser_open=keep_browser_open,
headless=headless,
disable_security=disable_security,
window_w=window_w,
window_h=window_h,
save_recording_path=save_recording_path,
save_trace_path=save_trace_path,
task=task,
max_steps=max_steps,
use_vision=use_vision,
max_actions_per_step=max_actions_per_step,
tool_call_in_content=tool_call_in_content
)
elif agent_type == "custom":
final_result, errors, model_actions, model_thoughts = await run_custom_agent(
llm=llm,
use_own_browser=use_own_browser,
browser_type=browser_type,
keep_browser_open=keep_browser_open,
headless=headless,
disable_security=disable_security,
window_w=window_w,
window_h=window_h,
save_recording_path=save_recording_path,
save_trace_path=save_trace_path,
task=task,
add_infos=add_infos,
max_steps=max_steps,
use_vision=use_vision,
max_actions_per_step=max_actions_per_step,
tool_call_in_content=tool_call_in_content
)
else:
raise ValueError(f"Invalid agent type: {agent_type}")
# Get the list of videos after the agent runs (if recording is enabled)
latest_video = None
if save_recording_path:
new_videos = set(
glob.glob(os.path.join(save_recording_path, "*.[mM][pP]4"))
+ glob.glob(os.path.join(save_recording_path, "*.[wW][eE][bB][mM]"))
)
if new_videos - existing_videos:
latest_video = list(new_videos - existing_videos)[0] # Get the first new video
return final_result, errors, model_actions, model_thoughts, latest_video
except asyncio.CancelledError:
return "Agent stopped", "", "Agent execution was stopped", "", None
async def run_org_agent(
llm,
use_own_browser,
browser_type,
keep_browser_open,
headless,
disable_security,
window_w,
window_h,
save_recording_path,
save_trace_path,
task,
max_steps,
use_vision,
max_actions_per_step,
tool_call_in_content
):
try:
global _global_browser, _global_browser_context
if use_own_browser:
chrome_path = os.getenv("CHROME_PATH", None)
if chrome_path == "":
chrome_path = None
else:
chrome_path = None
if _global_browser is None:
_global_browser = Browser(
config=BrowserConfig(
headless=headless,
disable_security=disable_security,
chrome_instance_path=chrome_path,
extra_chromium_args=[f"--window-size={window_w},{window_h}"],
)
)
if _global_browser_context is None:
_global_browser_context = await _global_browser.new_context(
config=BrowserContextConfig(
trace_path=save_trace_path if save_trace_path else None,
save_recording_path=save_recording_path if save_recording_path else None,
no_viewport=False,
browser_window_size=BrowserContextWindowSize(
width=window_w, height=window_h
),
)
)
agent = Agent(
task=task,
llm=llm,
use_vision=use_vision,
browser=_global_browser,
browser_context=_global_browser_context,
max_actions_per_step=max_actions_per_step,
tool_call_in_content=tool_call_in_content
)
history = await agent.run(max_steps=max_steps)
final_result = history.final_result()
errors = history.errors()
model_actions = history.model_actions()
model_thoughts = history.model_thoughts()
return final_result, errors, model_actions, model_thoughts
except Exception as e:
import traceback
traceback.print_exc()
errors = str(e) + "\n" + traceback.format_exc()
return '', errors, '', ''
finally:
# Handle cleanup based on persistence configuration
if not keep_browser_open:
if _global_browser_context:
await _global_browser_context.close()
_global_browser_context = None
if _global_browser:
await _global_browser.close()
_global_browser = None
async def run_custom_agent(
llm,
use_own_browser,
browser_type,
keep_browser_open,
headless,
disable_security,
window_w,
window_h,
save_recording_path,
save_trace_path,
task,
add_infos,
max_steps,
use_vision,
max_actions_per_step,
tool_call_in_content
):
try:
global _global_browser, _global_browser_context, _current_agent, _stop_event, _agent_state
if use_own_browser:
if browser_type == "Arc":
chrome_path = "/Applications/Arc.app/Contents/MacOS/Arc"
if not os.path.exists(chrome_path):
raise BrowserNotFoundError("Arc browser not found. Please install Arc or select Chrome.")
else:
chrome_path = os.getenv("CHROME_PATH", None)
if chrome_path == "":
chrome_path = None
else:
chrome_path = None
controller = CustomController()
# Initialize global browser if needed
if _global_browser is None:
_global_browser = CustomBrowser(
config=BrowserConfig(
headless=headless,
disable_security=disable_security,
chrome_instance_path=chrome_path,
extra_chromium_args=[f"--window-size={window_w},{window_h}"],
)
)
if _global_browser_context is None:
_global_browser_context = await _global_browser.new_context(
config=BrowserContextConfig(
trace_path=save_trace_path if save_trace_path else None,
save_recording_path=save_recording_path if save_recording_path else None,
no_viewport=False,
browser_window_size=BrowserContextWindowSize(
width=window_w, height=window_h
),
)
)
# Create and run agent
if _current_agent is None:
_current_agent = CustomAgent(
task=task,
add_infos=add_infos,
use_vision=use_vision,
llm=llm,
browser=_global_browser,
browser_context=_global_browser_context,
controller=controller,
system_prompt_class=CustomSystemPrompt,
max_actions_per_step=max_actions_per_step,
tool_call_in_content=tool_call_in_content,
stop_event=_stop_event
)
else:
# Update existing agent with new task and settings
_current_agent.task = task
_current_agent.add_infos = add_infos
_current_agent.use_vision = use_vision
_current_agent.llm = llm
_current_agent.max_actions_per_step = max_actions_per_step
_current_agent.tool_call_in_content = tool_call_in_content
_current_agent.stop_event = _stop_event
# Run the agent
history = await _current_agent.run(max_steps=max_steps)
final_result = history.final_result()
errors = history.errors()
model_actions = history.model_actions()
model_thoughts = history.model_thoughts()
return final_result, errors, model_actions, model_thoughts
except Exception as e:
import traceback
traceback.print_exc()
errors = str(e) + "\n" + traceback.format_exc()
return '', errors, '', ''
finally:
# Handle cleanup based on persistence configuration
if not keep_browser_open:
if _global_browser_context:
await _global_browser_context.close()
_global_browser_context = None
if _global_browser:
await _global_browser.close()
_global_browser = None
_current_agent = None # Clear the current agent if browser is closed
# Define the theme map globally
theme_map = {
"Default": Default(),
"Soft": Soft(),
"Monochrome": Monochrome(),
"Glass": Glass(),
"Base": Base()
}
async def close_global_browser():
global _global_browser, _global_browser_context
try:
if _global_browser_context:
await _global_browser_context.close()
_global_browser_context = None
if _global_browser:
await _global_browser.close()
_global_browser = None
except Exception as e:
logger.error(f"Error closing browser: {str(e)}")
def create_ui(theme_name="Default"):
css = """
.gradio-container {
max-width: 1200px !important;
margin: auto !important;
padding-top: 20px !important;
}
.header-text {
text-align: center;
margin-bottom: 30px;
}
.theme-section {
margin-bottom: 20px;
padding: 15px;
border-radius: 10px;
}
"""
js = """
function refresh() {
const url = new URL(window.location);
if (url.searchParams.get('__theme') !== 'dark') {
url.searchParams.set('__theme', 'dark');
window.location.href = url.href;
}
}
"""
with gr.Blocks(
title="Browser Use 100XPrompt", theme=theme_map[theme_name], css=css, js=js
) as demo:
with gr.Row():
gr.Markdown(
"""
# 🌐 100XPrompt
### Control your browser with 100X Prompt assistance
""",
elem_classes=["header-text"],
)
with gr.Tabs() as tabs:
with gr.TabItem("⚙️ Agent Settings", id=1):
with gr.Group():
agent_type = gr.Radio(
["org", "custom"],
label="Agent Type",
value="custom",
info="Select the type of agent to use",
)
max_steps = gr.Slider(
minimum=1,
maximum=200,
value=100,
step=1,
label="Max Run Steps",
info="Maximum number of steps the agent will take",
)
max_actions_per_step = gr.Slider(
minimum=1,
maximum=20,
value=10,
step=1,
label="Max Actions per Step",
info="Maximum number of actions the agent will take per step",
)
use_vision = gr.Checkbox(
label="Use Vision",
value=True,
info="Enable visual processing capabilities",
)
tool_call_in_content = gr.Checkbox(
label="Use Tool Calls in Content",
value=True,
info="Enable Tool Calls in content",
)
with gr.TabItem("🔧 LLM Configuration", id=2):
with gr.Group():
llm_provider = gr.Dropdown(
choices=[provider for provider,model in utils.model_names.items()],
label="LLM Provider",
value="openai",
info="Select your preferred language model provider"
)
llm_model_name = gr.Dropdown(
label="Model Name",
choices=utils.model_names['openai'],
value="gpt-4o",
interactive=True,
allow_custom_value=True, # Allow users to input custom model names
info="Select a model from the dropdown or type a custom model name"
)
llm_temperature = gr.Slider(
minimum=0.0,
maximum=2.0,
value=1.0,
step=0.1,
label="Temperature",
info="Controls randomness in model outputs"
)
with gr.Row():
llm_base_url = gr.Textbox(
label="Base URL",
value='',
info="API endpoint URL (if required)"
)
llm_api_key = gr.Textbox(
label="API Key",
type="password",
value='',
info="Your API key (leave blank to use .env)"
)
with gr.TabItem("🌐 Browser Settings", id=3):
with gr.Group():
with gr.Row():
browser_type = gr.Dropdown(
choices=["Chrome", "Arc"] if platform.system() == "Darwin" else ["Chrome"],
label="Browser Type",
value="Chrome",
info="Select browser to use (Arc available only on macOS)",
)
use_own_browser = gr.Checkbox(
label="Use Own Browser",
value=False,
info="Use your existing browser instance",
)
keep_browser_open = gr.Checkbox(
label="Keep Browser Open",
value=os.getenv("CHROME_PERSISTENT_SESSION", "False").lower() == "true",
info="Keep Browser Open between Tasks",
)
headless = gr.Checkbox(
label="Headless Mode",
value=False,
info="Run browser without GUI",
)
disable_security = gr.Checkbox(
label="Disable Security",
value=True,
info="Disable browser security features",
)
enable_recording = gr.Checkbox(
label="Enable Recording",
value=True,
info="Enable saving browser recordings",
)
with gr.Row():
window_w = gr.Number(
label="Window Width",
value=1280,
info="Browser window width",
)
window_h = gr.Number(
label="Window Height",
value=1100,
info="Browser window height",
)
save_recording_path = gr.Textbox(
label="Recording Path",
placeholder="e.g. ./tmp/record_videos",
value="./tmp/record_videos",
info="Path to save browser recordings",
interactive=True, # Allow editing only if recording is enabled
)
save_trace_path = gr.Textbox(
label="Trace Path",
placeholder="e.g. ./tmp/traces",
value="./tmp/traces",
info="Path to save Agent traces",
interactive=True,
)
with gr.TabItem("🤖 Run Agent", id=4):
task = gr.Textbox(
label="Task Description",
lines=4,
placeholder="Enter your task here...",
value="go to google.com and type 'OpenAI' click search and give me the first url",
info="Describe what you want the agent to do",
)
add_infos = gr.Textbox(
label="Additional Information",
lines=3,
placeholder="Add any helpful context or instructions...",
info="Optional hints to help the LLM complete the task",
)
# Define output components first
with gr.Group():
gr.Markdown("### Results")
with gr.Row():
with gr.Column():
final_result_output = gr.Textbox(
label="Final Result", lines=3, show_label=True
)
with gr.Column():
errors_output = gr.Textbox(
label="Errors", lines=3, show_label=True
)
with gr.Row():
with gr.Column():
model_actions_output = gr.Textbox(
label="Model Actions", lines=3, show_label=True
)
with gr.Column():
model_thoughts_output = gr.Textbox(
label="Model Thoughts", lines=3, show_label=True
)
recording_display = gr.Video(label="Latest Recording")
with gr.Row():
run_button = gr.Button("▶️ Run Agent", variant="primary", scale=2)
stop_button = gr.Button("⏹️ Stop", variant="stop", scale=1)
resume_button = gr.Button("⏵️ Resume", variant="secondary", scale=1)
# Add event handlers for stop and resume
stop_button.click(
fn=stop_agent,
inputs=[],
outputs=[],
)
resume_button.click(
fn=resume_agent,
inputs=[task],
outputs=[],
).then(
fn=run_browser_agent,
inputs=[
agent_type, llm_provider, llm_model_name, llm_temperature, llm_base_url, llm_api_key,
use_own_browser, browser_type, keep_browser_open, headless, disable_security, window_w, window_h,
save_recording_path, save_trace_path, enable_recording, task, add_infos, max_steps,
use_vision, max_actions_per_step, tool_call_in_content
],
outputs=[final_result_output, errors_output, model_actions_output, model_thoughts_output, recording_display],
)
# Add run button handler
run_button.click(
fn=run_browser_agent,
inputs=[
agent_type, llm_provider, llm_model_name, llm_temperature, llm_base_url, llm_api_key,
use_own_browser, browser_type, keep_browser_open, headless, disable_security, window_w, window_h,
save_recording_path, save_trace_path, enable_recording, task, add_infos, max_steps,
use_vision, max_actions_per_step, tool_call_in_content
],
outputs=[final_result_output, errors_output, model_actions_output, model_thoughts_output, recording_display],
)
with gr.TabItem("🎥 Recordings", id=6):
def list_recordings(save_recording_path):
if not os.path.exists(save_recording_path):
return []
# Get all video files
recordings = glob.glob(os.path.join(save_recording_path, "*.[mM][pP]4")) + glob.glob(os.path.join(save_recording_path, "*.[wW][eE][bB][mM]"))
# Sort recordings by creation time (oldest first)
recordings.sort(key=os.path.getctime)
# Add numbering to the recordings
numbered_recordings = []
for idx, recording in enumerate(recordings, start=1):
filename = os.path.basename(recording)
numbered_recordings.append((recording, f"{idx}. {filename}"))
return numbered_recordings
recordings_gallery = gr.Gallery(
label="Recordings",
value=list_recordings("./tmp/record_videos"),
columns=3,
height="auto",
object_fit="contain"
)
refresh_button = gr.Button("🔄 Refresh Recordings", variant="secondary")
refresh_button.click(
fn=list_recordings,
inputs=save_recording_path,
outputs=recordings_gallery
)
# Attach the callback to the LLM provider dropdown
llm_provider.change(
lambda provider, api_key, base_url: update_model_dropdown(provider, api_key, base_url),
inputs=[llm_provider, llm_api_key, llm_base_url],
outputs=llm_model_name
)
# Add this after defining the components
enable_recording.change(
lambda enabled: gr.update(interactive=enabled),
inputs=enable_recording,
outputs=save_recording_path
)
# Add the change handler here, inside the Blocks context
def update_browser_type(use_own):
return gr.update(interactive=use_own)
use_own_browser.change(
fn=update_browser_type,
inputs=use_own_browser,
outputs=browser_type
)
return demo
def main():
parser = argparse.ArgumentParser(description="Gradio UI for Browser Agent")
parser.add_argument("--ip", type=str, default="0.0.0.0", help="IP address to bind to") # Change to 0.0.0.0
parser.add_argument("--port", type=int, default=7788, help="Port to listen on")
parser.add_argument("--theme", type=str, default="Default", choices=list(theme_map.keys()))
parser.add_argument("--share", action="store_true", help="Enable sharing of the UI")
args = parser.parse_args()
demo = create_ui(theme_name=args.theme)
demo.launch(server_name="0.0.0.0", server_port=7788, share=True) # Add share=True here
if __name__ == '__main__':
main()