Lancer / app /agents /browser_tools.py
Madras1's picture
Upload 89 files
4bcc05b verified
Raw
History Blame Contribute Delete
7.95 kB
"""Reusable tooling for browser-agent action selection.
This module gives the browser agents a shared, validated action surface.
It supports both:
- native tool calling when the LLM/provider can emit tool calls
- JSON fallback when the model only returns plain text
"""
from __future__ import annotations
import json
from typing import Any
from app.agents.tooling import ToolCall, ToolRegistry, tool
def _normalize_points(values: list[str] | str | None) -> list[str]:
"""Trim and deduplicate short research memory lists.
Models occasionally return a plain string instead of a list. Treat that as a
single item instead of iterating character-by-character.
"""
cleaned: list[str] = []
if values is None:
iterable: list[Any] = []
elif isinstance(values, str):
iterable = [values]
else:
iterable = list(values)
for value in iterable:
text = str(value).strip()
if len(text) <= 1 and text.isalpha():
continue
if text and text not in cleaned:
cleaned.append(text)
return cleaned
@tool(description="Search the web with a fresh query when the current page is insufficient.")
def search_web(
query: str,
reason: str = "",
known_facts: list[str] | None = None,
missing_points: list[str] | None = None,
) -> dict[str, Any]:
"""Search the web.
Args:
query: Query terms to search for next.
reason: Why a new search is needed.
known_facts: Short facts already established.
missing_points: What information is still missing.
"""
return {
"action": "SEARCH",
"value": query.strip(),
"reason": reason.strip(),
"known_facts": _normalize_points(known_facts),
"missing_points": _normalize_points(missing_points),
}
@tool(description="Open a new URL that was discovered in the current page or results.")
def navigate_to_url(
url: str,
reason: str = "",
known_facts: list[str] | None = None,
missing_points: list[str] | None = None,
) -> dict[str, Any]:
"""Navigate to a specific URL.
Args:
url: Absolute URL to visit next.
reason: Why that URL is the best next step.
known_facts: Short facts already established.
missing_points: What information is still missing.
"""
return {
"action": "NAVIGATE",
"value": url.strip(),
"reason": reason.strip(),
"known_facts": _normalize_points(known_facts),
"missing_points": _normalize_points(missing_points),
}
@tool(description="Scroll the current page to reveal more content.")
def scroll_page(
reason: str = "",
known_facts: list[str] | None = None,
missing_points: list[str] | None = None,
) -> dict[str, Any]:
"""Scroll the current page.
Args:
reason: Why scrolling is useful right now.
known_facts: Short facts already established.
missing_points: What information is still missing.
"""
return {
"action": "SCROLL",
"value": "",
"reason": reason.strip(),
"known_facts": _normalize_points(known_facts),
"missing_points": _normalize_points(missing_points),
}
@tool(description="Finish the task and provide the final answer based on the collected evidence.")
def finish_task(
answer: str,
reason: str = "",
known_facts: list[str] | None = None,
missing_points: list[str] | None = None,
) -> dict[str, Any]:
"""Finish the task.
Args:
answer: Final user-facing answer.
reason: Why the task is complete.
known_facts: Short facts already established.
missing_points: Remaining uncertainty, if any.
"""
return {
"action": "DONE",
"value": "",
"answer": answer.strip(),
"reason": reason.strip(),
"known_facts": _normalize_points(known_facts),
"missing_points": _normalize_points(missing_points),
}
BROWSER_TOOL_REGISTRY = ToolRegistry([
search_web,
navigate_to_url,
scroll_page,
finish_task,
])
def get_browser_tools(allow_scroll: bool = True) -> list[dict[str, Any]]:
"""Return OpenAI-compatible tool schemas for the browser agent."""
tools = []
for schema in BROWSER_TOOL_REGISTRY.schemas:
if not allow_scroll and schema.name == "scroll_page":
continue
tools.append(schema.to_openai_tool())
return tools
def execute_browser_tool_call(tool_call: ToolCall, allow_scroll: bool = True) -> dict[str, Any]:
"""Execute a browser decision tool call and validate mode-specific constraints."""
if not allow_scroll and tool_call.name == "scroll_page":
raise ValueError("scroll_page is not allowed for this browser mode")
result = BROWSER_TOOL_REGISTRY.execute(tool_call)
return validate_browser_decision(result, allow_scroll=allow_scroll)
def parse_browser_json_response(text: str, allow_scroll: bool = True) -> dict[str, Any]:
"""Parse legacy JSON action output into the normalized browser-decision shape."""
snippet = _extract_json_object(text)
data = json.loads(snippet)
action = str(data.get("action", "DONE")).strip().upper()
normalized = {
"action": action,
"value": str(data.get("value", "")).strip(),
"answer": str(data.get("answer", "")).strip(),
"reason": str(data.get("reason", "")).strip(),
"known_facts": _normalize_points(data.get("known_facts")),
"missing_points": _normalize_points(data.get("missing_points")),
}
if action == "SEARCH":
normalized["value"] = str(data.get("query", normalized["value"])).strip()
elif action == "NAVIGATE":
normalized["value"] = str(data.get("url", normalized["value"])).strip()
elif action == "DONE":
normalized["answer"] = str(data.get("answer", data.get("result", normalized["answer"]))).strip()
elif action == "SCROLL":
normalized["value"] = ""
return validate_browser_decision(normalized, allow_scroll=allow_scroll)
def validate_browser_decision(decision: dict[str, Any], allow_scroll: bool = True) -> dict[str, Any]:
"""Validate and normalize a browser agent decision."""
action = str(decision.get("action", "")).strip().upper()
value = str(decision.get("value", "")).strip()
answer = str(decision.get("answer", "")).strip()
normalized = {
"action": action or "DONE",
"value": value,
"answer": answer,
"reason": str(decision.get("reason", "")).strip(),
"known_facts": _normalize_points(decision.get("known_facts")),
"missing_points": _normalize_points(decision.get("missing_points")),
}
if normalized["action"] == "SEARCH":
if not normalized["value"]:
raise ValueError("SEARCH decision requires a non-empty query")
elif normalized["action"] == "NAVIGATE":
if not normalized["value"].startswith("http"):
raise ValueError("NAVIGATE decision requires an absolute URL")
elif normalized["action"] == "SCROLL":
if not allow_scroll:
raise ValueError("SCROLL is not supported in this browser mode")
normalized["value"] = ""
elif normalized["action"] == "DONE":
pass
else:
raise ValueError(f"Unsupported browser action '{normalized['action']}'")
return normalized
def _extract_json_object(text: str) -> str:
"""Extract the first JSON object-looking slice from a model response."""
raw = (text or "").strip()
if raw.startswith("```"):
parts = raw.split("```")
if len(parts) >= 2:
raw = parts[1]
if raw.startswith("json"):
raw = raw[4:]
raw = raw.strip()
start = raw.find("{")
end = raw.rfind("}")
if start == -1 or end == -1 or end <= start:
raise ValueError("Model response did not contain a JSON object")
return raw[start:end + 1]