emo-online-api / emo /backend /web_tools.py
Emo Online
Deploy Emo API
dd87944
Raw
History Blame Contribute Delete
23.6 kB
"""Web access tools for Émo — multi-source search (style Cursor) + fetch."""
import ssl_fix # noqa: F401
import asyncio
import ast
import httpx
import json
import logging
import math
import operator
import re
from datetime import datetime, timezone
from typing import Optional
from urllib.parse import quote_plus, urljoin, urlparse
from bs4 import BeautifulSoup
logger = logging.getLogger("emo.web")
USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
FOCUS_SUFFIX = {
"code": ["site:stackoverflow.com", "site:github.com"],
"docs": ["documentation", "site:readthedocs.io"],
"news": ["news"],
}
def _domain(url: str) -> str:
try:
return urlparse(url).netloc.replace("www.", "")
except Exception:
return ""
# Sites qui bloquent le fetch serveur ou l'embed — aperçu carte + lien externe uniquement.
_EXTERNAL_ONLY = re.compile(
r"(^|\.)("
r"youtube\.com|youtu\.be|facebook\.com|instagram\.com|twitter\.com|x\.com|"
r"tiktok\.com|linkedin\.com|netflix\.com|spotify\.com|accounts\.google\.com|"
r"discord\.com|twitch\.tv|kick\.com"
r")$",
re.I,
)
_SITE_TITLES = {
"youtube.com": "YouTube",
"youtu.be": "YouTube",
"google.com": "Google",
"facebook.com": "Facebook",
"instagram.com": "Instagram",
"twitter.com": "X (Twitter)",
"x.com": "X (Twitter)",
"tiktok.com": "TikTok",
"netflix.com": "Netflix",
"spotify.com": "Spotify",
"discord.com": "Discord",
"twitch.tv": "Twitch",
"kick.com": "Kick",
}
def _external_only_site(url: str) -> bool:
host = _domain(url)
if not host:
return False
return bool(_EXTERNAL_ONLY.search(host))
def _site_title(url: str) -> str:
host = _domain(url)
if host in _SITE_TITLES:
return _SITE_TITLES[host]
for key, title in _SITE_TITLES.items():
if host.endswith("." + key) or host == key:
return title
return host.split(".")[0].capitalize() if host else "Site web"
def _youtube_video_id(url: str) -> str | None:
try:
from urllib.parse import urlparse, parse_qs
p = urlparse(url)
host = (p.netloc or "").lower()
if "youtu.be" in host:
vid = p.path.lstrip("/").split("/")[0]
return vid or None
if "youtube.com" in host:
if p.path == "/watch":
return (parse_qs(p.query).get("v") or [None])[0]
m_live = re.match(r"^/live/([^/?]+)", p.path)
if m_live:
return m_live.group(1)
m = re.match(r"^/(embed|shorts|v)/([^/?]+)", p.path)
if m:
return m.group(2)
except Exception:
pass
return None
def _video_embed_url(url: str, parent: str = "localhost") -> str | None:
"""URL lecteur embed pour vidéos / lives (YouTube, Twitch, Kick)."""
vid = _youtube_video_id(url)
if vid:
return f"https://www.youtube-nocookie.com/embed/{vid}?rel=0&modestbranding=1&playsinline=1"
try:
p = urlparse(url)
host = (p.netloc or "").lower()
parts = [x for x in p.path.split("/") if x]
if "twitch.tv" in host:
parent_q = quote_plus(parent or "localhost")
if parts and parts[0] == "videos" and len(parts) > 1:
return f"https://player.twitch.tv/?video={quote_plus(parts[1])}&parent={parent_q}&autoplay=true"
channel = parts[0] if parts else ""
if channel and channel.lower() not in ("directory", "settings", "downloads", "p", "search"):
return f"https://player.twitch.tv/?channel={quote_plus(channel)}&parent={parent_q}&autoplay=true"
if "kick.com" in host:
channel = parts[0] if parts else ""
if channel and channel.lower() not in ("categories", "browse", "search"):
return f"https://player.kick.com/{quote_plus(channel)}"
except Exception:
pass
return None
def _synthetic_browser_visit(url: str, *, note: str = "") -> dict:
"""Aperçu fiable quand le fetch HTTP échoue (YouTube, réseaux sociaux, etc.)."""
url = normalize_url(url)
title = _site_title(url)
host = _domain(url)
favicon = f"https://www.google.com/s2/favicons?domain={host}&sz=128" if host else ""
images = [{"url": favicon, "alt": title}] if favicon else []
vid = _youtube_video_id(url)
embed = _video_embed_url(url)
if embed:
preview = f"{title} — vidéo prête à lire dans le navigateur intégré."
else:
preview = (
f"{title} — ce site bloque la lecture automatique depuis le serveur. "
f"Utilise le bouton **Ouvrir** pour l'afficher dans ton navigateur."
)
out = {
"ok": True,
"url": url,
"title": title,
"text": preview,
"preview": preview,
"links": [{"text": f"Ouvrir {title}", "url": url}],
"images": images,
"embed_blocked": not bool(embed),
"hint": "Site ouvert en mode aperçu externe." if not embed else "Vidéo — lecteur embed dans le panneau.",
}
if vid:
out["youtube_video_id"] = vid
if embed:
out["embed_url"] = embed
if note:
out["fetch_note"] = note[:200]
return out
def normalize_url(url: str) -> str:
"""Préfixe https:// si absent — évite les échecs browser_visit / web_fetch."""
u = (url or "").strip()
if not u:
return u
if u.startswith("data:"):
return u
if not u.startswith(("http://", "https://")):
u = f"https://{u}"
return u
def _dedupe_results(items: list[dict], limit: int) -> list[dict]:
seen: set[str] = set()
out: list[dict] = []
for item in items:
url = (item.get("url") or "").split("#")[0].rstrip("/")
if not url or url in seen:
continue
seen.add(url)
item["domain"] = _domain(url)
out.append(item)
if len(out) >= limit:
break
return out
async def _search_ddg(client: httpx.AsyncClient, query: str, limit: int) -> list[dict]:
url = f"https://html.duckduckgo.com/html/?q={quote_plus(query)}"
r = await client.get(url)
if r.status_code != 200:
return []
soup = BeautifulSoup(r.text, "lxml")
results = []
for el in soup.select("div.result")[: limit + 4]:
title_a = el.select_one("a.result__a")
snippet_el = el.select_one(".result__snippet")
if not title_a:
continue
link = title_a.get("href", "")
m = re.search(r"uddg=([^&]+)", link)
if m:
from urllib.parse import unquote
link = unquote(m.group(1))
results.append({
"title": title_a.get_text(strip=True),
"url": link,
"snippet": snippet_el.get_text(strip=True) if snippet_el else "",
"source": "duckduckgo",
})
return results
async def _search_bing(client: httpx.AsyncClient, query: str, limit: int) -> list[dict]:
url = f"https://www.bing.com/search?q={quote_plus(query)}"
r = await client.get(url)
if r.status_code != 200:
return []
soup = BeautifulSoup(r.text, "lxml")
results = []
for li in soup.select("li.b_algo")[: limit + 4]:
a = li.select_one("h2 a")
if not a:
continue
link = a.get("href", "")
snippet_el = li.select_one(".b_caption p") or li.select_one("p")
results.append({
"title": a.get_text(strip=True),
"url": link,
"snippet": snippet_el.get_text(strip=True) if snippet_el else "",
"source": "bing",
})
return results
def _expand_queries(query: str, focus: str) -> list[str]:
q = query.strip()
if not q:
return []
queries = [q]
focus = (focus or "general").lower()
for suffix in FOCUS_SUFFIX.get(focus, []):
variant = f"{q} {suffix}".strip()
if variant not in queries:
queries.append(variant)
return queries[:3]
async def web_search(
query: str,
limit: int = 10,
focus: str = "general",
queries: list | None = None,
) -> dict:
"""Multi-source web search with focus modes (Cursor-style research)."""
search_queries = queries if queries else _expand_queries(query, focus)
if not search_queries:
return {"ok": False, "error": "query missing"}
limit = max(1, min(int(limit or 10), 20))
try:
async with httpx.AsyncClient(
timeout=22,
headers={"User-Agent": USER_AGENT, "Accept-Language": "fr-FR,fr;q=0.9,en;q=0.8"},
follow_redirects=True,
) as client:
merged: list[dict] = []
for sq in search_queries:
ddg, bing = await asyncio.gather(
_search_ddg(client, sq, limit),
_search_bing(client, sq, limit),
return_exceptions=True,
)
if isinstance(ddg, list):
merged.extend(ddg)
if isinstance(bing, list):
merged.extend(bing)
if len(_dedupe_results(merged, limit + 5)) >= limit:
break
results = _dedupe_results(merged, limit)
for i, r in enumerate(results, 1):
r["rank"] = i
return {
"ok": True,
"query": query,
"focus": focus,
"queries_run": search_queries,
"results": results,
"count": len(results),
"hint": "Utilise web_fetch sur les 1-2 URLs les plus pertinentes avant de répondre.",
}
except Exception as e:
logger.exception("web_search failed")
return {"ok": False, "error": str(e)}
async def web_fetch(url: str, max_chars: int = 12000) -> dict:
"""Fetch a URL and return clean readable text + links + title."""
url = normalize_url(url)
if not url or not url.startswith(("http://", "https://")):
return {"ok": False, "error": "URL invalide (doit commencer par http:// ou https://)"}
try:
async with httpx.AsyncClient(
timeout=25,
headers={"User-Agent": USER_AGENT, "Accept-Language": "fr-FR,fr;q=0.9,en;q=0.8"},
follow_redirects=True,
) as client:
r = await client.get(url)
if r.status_code >= 400:
return {"ok": False, "error": f"HTTP {r.status_code}", "url": url}
ctype = r.headers.get("content-type", "").lower()
if "text/html" not in ctype and "application/xhtml" not in ctype:
text = r.text[:max_chars] if hasattr(r, "text") else ""
return {
"ok": True, "url": str(r.url), "title": "",
"content_type": ctype, "text": text, "links": [], "images": [],
}
soup = BeautifulSoup(r.text, "lxml")
for tag in soup(["script", "style", "noscript", "iframe", "svg"]):
tag.decompose()
title = (soup.title.get_text(strip=True) if soup.title else "")[:200]
main = soup.find("main") or soup.find("article") or soup.body or soup
text = main.get_text("\n", strip=True)
text = re.sub(r"\n{3,}", "\n\n", text)[:max_chars]
base = str(r.url)
links = []
for a in soup.select("a[href]")[:40]:
href = a.get("href", "").strip()
if not href or href.startswith("#"):
continue
absolute = urljoin(base, href)
label = a.get_text(strip=True)[:80]
if label and absolute.startswith(("http://", "https://")):
links.append({"text": label, "url": absolute})
images = []
for img in soup.select("img[src]")[:20]:
src = img.get("src", "").strip()
if src:
absolute = urljoin(base, src)
if absolute.startswith(("http://", "https://")):
images.append({"url": absolute, "alt": img.get("alt", "")[:80]})
return {
"ok": True,
"url": str(r.url),
"title": title,
"content_type": ctype,
"text": text,
"links": links[:25],
"images": images,
"truncated": len(text) >= max_chars,
}
except Exception as e:
logger.exception("web_fetch failed")
return {"ok": False, "error": str(e), "url": url}
async def get_datetime(tz_hint: str = "UTC") -> dict:
now = datetime.now(timezone.utc)
return {
"ok": True,
"iso": now.isoformat(),
"unix": int(now.timestamp()),
"timezone": tz_hint or "UTC",
"weekday": now.strftime("%A"),
}
async def web_fetch_json(url: str, max_chars: int = 8000) -> dict:
if not url or not url.startswith(("http://", "https://")):
return {"ok": False, "error": "URL invalide"}
try:
async with httpx.AsyncClient(timeout=20, headers={"User-Agent": USER_AGENT}, follow_redirects=True) as client:
r = await client.get(url)
if r.status_code >= 400:
return {"ok": False, "error": f"HTTP {r.status_code}", "url": url}
text = r.text[:max_chars]
try:
data = json.loads(text)
return {"ok": True, "url": str(r.url), "json": data}
except json.JSONDecodeError:
return {"ok": True, "url": str(r.url), "text": text, "note": "not valid JSON"}
except Exception as e:
return {"ok": False, "error": str(e), "url": url}
async def github_search(query: str, limit: int = 8, access_token: Optional[str] = None) -> dict:
q = (query or "").strip()
if not q:
return {"ok": False, "error": "query missing"}
if access_token:
from connected_accounts import github_search_authenticated
return await github_search_authenticated(q, access_token, limit=min(limit, 15))
return await web_search(q, limit=min(limit, 15), focus="code")
async def github_api(
access_token: str,
method: str,
path: str,
*,
params: Optional[dict] = None,
json_body: Optional[dict] = None,
) -> dict:
from connected_accounts import github_api as _github_api
return await _github_api(access_token, method, path, params=params, json_body=json_body)
async def stackoverflow_search(query: str, limit: int = 8) -> dict:
q = (query or "").strip()
if not q:
return {"ok": False, "error": "query missing"}
return await web_search(f"{q} site:stackoverflow.com", limit=min(limit, 15), focus="code")
_SAFE_OPS = {
ast.Add: operator.add, ast.Sub: operator.sub, ast.Mult: operator.mul,
ast.Div: operator.truediv, ast.Pow: operator.pow, ast.Mod: operator.mod,
ast.USub: operator.neg, ast.UAdd: operator.pos,
}
def _safe_eval(node):
if isinstance(node, ast.Constant) and isinstance(node.value, (int, float)):
return node.value
if isinstance(node, ast.BinOp) and type(node.op) in _SAFE_OPS:
return _SAFE_OPS[type(node.op)](_safe_eval(node.left), _safe_eval(node.right))
if isinstance(node, ast.UnaryOp) and type(node.op) in _SAFE_OPS:
return _SAFE_OPS[type(node.op)](_safe_eval(node.operand))
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name):
fn = node.func.id
args = [_safe_eval(a) for a in node.args]
if fn == "sqrt" and len(args) == 1:
return math.sqrt(args[0])
if fn == "abs" and len(args) == 1:
return abs(args[0])
if fn == "round" and 1 <= len(args) <= 2:
return round(*args)
raise ValueError("expression non supportée")
def calculate_expression(expression: str) -> dict:
expr = (expression or "").strip()
if not expr:
return {"ok": False, "error": "expression missing"}
if len(expr) > 200:
return {"ok": False, "error": "expression trop longue"}
try:
tree = ast.parse(expr, mode="eval")
val = _safe_eval(tree.body)
return {"ok": True, "expression": expr, "result": val}
except Exception as e:
return {"ok": False, "error": str(e), "expression": expr}
async def browser_visit(url: str, max_chars: int = 10000) -> dict:
"""Ouvre une page web (texte + liens) — visible dans le panneau Navigateur de l'UI."""
url = normalize_url(url)
if not url or not url.startswith(("http://", "https://")):
return {"ok": False, "error": "URL invalide"}
if _external_only_site(url):
return _synthetic_browser_visit(url)
result = await web_fetch(url, max_chars=max_chars)
if not result.get("ok"):
err = result.get("error") or "fetch impossible"
if _external_only_site(url):
return _synthetic_browser_visit(url, note=err)
return {**result, "error": err}
text = result.get("text") or ""
return {
**result,
"preview": text[:2000],
"links_count": len(result.get("links") or []),
"images_count": len(result.get("images") or []),
"hint": "Contenu lu — cite l'URL dans ta réponse.",
}
WEB_TOOLS = [
{
"type": "function",
"function": {
"name": "web_search",
"description": (
"Recherche web multi-sources (DuckDuckGo + Bing), style Cursor. "
"Retourne title, url, snippet, domain, rank. "
"focus=code pour Stack Overflow/GitHub, docs pour documentation, news pour actualités. "
"Enchaîne avec web_fetch sur les meilleurs résultats."
),
"parameters": {
"type": "object",
"properties": {
"query": {"type": "string", "description": "Requête principale."},
"limit": {"type": "integer", "description": "Nb max de résultats (défaut 10, max 20)."},
"focus": {
"type": "string",
"enum": ["general", "code", "docs", "news"],
"description": "Type de recherche (défaut general).",
},
"queries": {
"type": "array",
"items": {"type": "string"},
"description": "Variantes de requêtes optionnelles (recherches parallèles).",
},
},
"required": ["query"],
},
},
},
{
"type": "function",
"function": {
"name": "browser_visit",
"description": (
"Ouvre une page web dans le navigateur d'Émo (panneau Activité + aperçu inline dans le chat). "
"OBLIGATOIRE quand l'utilisateur demande d'ouvrir/afficher un site dans le chat. "
"Lit le contenu, extrait titre, texte, liens."
),
"parameters": {
"type": "object",
"properties": {
"url": {"type": "string", "description": "URL complète https://…"},
"max_chars": {"type": "integer", "description": "Limite texte extrait (défaut 10000)."},
},
"required": ["url"],
},
},
},
{
"type": "function",
"function": {
"name": "web_fetch",
"description": "Récupère le contenu (texte propre + liens + images URLs) d'une page web. Utilise après web_search pour lire doc, README GitHub, Stack Overflow, etc.",
"parameters": {
"type": "object",
"properties": {
"url": {"type": "string", "description": "URL complète (http:// ou https://)."},
"max_chars": {"type": "integer", "description": "Limite de caractères du texte extrait (défaut 12000)."},
},
"required": ["url"],
},
},
},
{
"type": "function",
"function": {
"name": "web_fetch_json",
"description": "Récupère une URL JSON (API REST, GitHub raw, etc.) et parse le JSON.",
"parameters": {
"type": "object",
"properties": {
"url": {"type": "string", "description": "URL JSON."},
"max_chars": {"type": "integer", "description": "Limite taille réponse."},
},
"required": ["url"],
},
},
},
{
"type": "function",
"function": {
"name": "get_datetime",
"description": "Date/heure UTC actuelle (pour deadlines, logs, planification).",
"parameters": {
"type": "object",
"properties": {
"timezone": {"type": "string", "description": "Indication fuseau (info seulement)."},
},
},
},
},
{
"type": "function",
"function": {
"name": "github_search",
"description": (
"Recherche GitHub (repos, issues, code). "
"Si l'utilisateur a lié son compte GitHub, utilise l'API authentifiée."
),
"parameters": {
"type": "object",
"properties": {
"query": {"type": "string", "description": "Requête GitHub."},
"limit": {"type": "integer", "description": "Max résultats."},
},
"required": ["query"],
},
},
},
{
"type": "function",
"function": {
"name": "github_api",
"description": (
"Appel authentifié à l'API GitHub REST (repos, issues, PRs, gists…). "
"Nécessite un compte GitHub connecté par l'utilisateur."
),
"parameters": {
"type": "object",
"properties": {
"method": {"type": "string", "enum": ["GET", "POST", "PATCH", "PUT", "DELETE"], "description": "Verbe HTTP."},
"path": {"type": "string", "description": "Chemin API (ex: repos/owner/repo/issues) ou URL complète."},
"params": {"type": "object", "description": "Query params optionnels."},
"json": {"type": "object", "description": "Corps JSON pour POST/PATCH."},
},
"required": ["path"],
},
},
},
{
"type": "function",
"function": {
"name": "stackoverflow_search",
"description": "Recherche Stack Overflow pour bugs et solutions code.",
"parameters": {
"type": "object",
"properties": {
"query": {"type": "string", "description": "Question / erreur."},
"limit": {"type": "integer", "description": "Max résultats."},
},
"required": ["query"],
},
},
},
{
"type": "function",
"function": {
"name": "calculate",
"description": "Calculatrice sûre (+ - * / ^ sqrt abs round). Pas de code arbitraire.",
"parameters": {
"type": "object",
"properties": {
"expression": {"type": "string", "description": "Ex: (42 * 1.2) + sqrt(16)"},
},
"required": ["expression"],
},
},
},
]