Spaces:
Running on Zero
Running on Zero
Improve search: add news/images backends, page extraction, region support
Browse files
app.py
CHANGED
|
@@ -148,14 +148,17 @@ API_RESOURCES = {
|
|
| 148 |
},
|
| 149 |
"web_search": {
|
| 150 |
"name": "Web search",
|
| 151 |
-
"description": "Search the web via DuckDuckGo. No API key required.",
|
| 152 |
"endpoint": "/respite/search",
|
| 153 |
"method": "GET",
|
| 154 |
"input": {
|
| 155 |
"q": "string (search query)",
|
|
|
|
| 156 |
"max_results": "integer (1-50, default 10)",
|
|
|
|
|
|
|
| 157 |
},
|
| 158 |
-
"output": "JSON with title, url, content per result",
|
| 159 |
}
|
| 160 |
}
|
| 161 |
|
|
@@ -211,19 +214,78 @@ def _get_runtime_specs():
|
|
| 211 |
try:
|
| 212 |
from ddgs import DDGS
|
| 213 |
|
| 214 |
-
def _search_web(query, max_results=10):
|
| 215 |
-
"""Search the web via DuckDuckGo.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 216 |
with DDGS() as ddgs:
|
| 217 |
-
|
| 218 |
-
|
| 219 |
-
|
| 220 |
-
|
| 221 |
-
|
| 222 |
-
|
| 223 |
-
|
| 224 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 225 |
return {
|
| 226 |
"query": query,
|
|
|
|
| 227 |
"results": results,
|
| 228 |
"total_results": len(results),
|
| 229 |
}
|
|
@@ -231,7 +293,7 @@ try:
|
|
| 231 |
except ImportError:
|
| 232 |
_log("Warning: duckduckgo_search not installed, search disabled")
|
| 233 |
|
| 234 |
-
def _search_web(query, max_results=10):
|
| 235 |
raise RuntimeError("Search not available: duckduckgo_search not installed")
|
| 236 |
|
| 237 |
|
|
@@ -336,11 +398,13 @@ def _patched_create_app(blocks, **kwargs):
|
|
| 336 |
return JSONResponse({**API_SPECS, "runtime": _get_runtime_specs()})
|
| 337 |
|
| 338 |
@fa_app.get("/respite/search")
|
| 339 |
-
def _search(q: str = "", max_results: int = 10):
|
| 340 |
if not q:
|
| 341 |
return JSONResponse({"error": "Missing query parameter 'q'"}, status_code=400)
|
|
|
|
|
|
|
| 342 |
try:
|
| 343 |
-
results = _search_web(q, max_results=max_results)
|
| 344 |
return JSONResponse(results)
|
| 345 |
except Exception as e:
|
| 346 |
return JSONResponse({"error": str(e)}, status_code=502)
|
|
|
|
| 148 |
},
|
| 149 |
"web_search": {
|
| 150 |
"name": "Web search",
|
| 151 |
+
"description": "Search the web via DuckDuckGo. Supports general, news, and image search. No API key required.",
|
| 152 |
"endpoint": "/respite/search",
|
| 153 |
"method": "GET",
|
| 154 |
"input": {
|
| 155 |
"q": "string (search query)",
|
| 156 |
+
"backend": "string: 'text' (default), 'news', or 'images'",
|
| 157 |
"max_results": "integer (1-50, default 10)",
|
| 158 |
+
"extract_top": "integer (0-5, default 0) - fetch full page content for top N results",
|
| 159 |
+
"region": "string (default 'wt-wt') - e.g. 'us-en', 'gb-en', 'de-de'",
|
| 160 |
},
|
| 161 |
+
"output": "JSON with title, url, content per result. News results include date and source. Extracted results include extracted_content.",
|
| 162 |
}
|
| 163 |
}
|
| 164 |
|
|
|
|
| 214 |
try:
|
| 215 |
from ddgs import DDGS
|
| 216 |
|
| 217 |
+
def _search_web(query, backend="text", max_results=10, extract_top=0, region="wt-wt"):
|
| 218 |
+
"""Search the web via DuckDuckGo.
|
| 219 |
+
|
| 220 |
+
Args:
|
| 221 |
+
query: Search query string.
|
| 222 |
+
backend: "text" (general), "news" (news articles), or "images".
|
| 223 |
+
max_results: Number of results to return (1-50).
|
| 224 |
+
extract_top: Fetch full page content for top N results (0-5).
|
| 225 |
+
region: DuckDuckGo region code (e.g. "us-en", "gb-en", "wt-wt" for worldwide).
|
| 226 |
+
"""
|
| 227 |
+
extract_top = max(0, min(5, extract_top))
|
| 228 |
+
max_results = max(1, min(50, max_results))
|
| 229 |
+
|
| 230 |
with DDGS() as ddgs:
|
| 231 |
+
if backend == "news":
|
| 232 |
+
hits = list(ddgs.news(query, max_results=max_results, region=region))
|
| 233 |
+
results = []
|
| 234 |
+
for r in hits:
|
| 235 |
+
entry = {
|
| 236 |
+
"title": r.get("title", ""),
|
| 237 |
+
"url": r.get("url", ""),
|
| 238 |
+
"content": r.get("body", ""),
|
| 239 |
+
"source": r.get("source", ""),
|
| 240 |
+
"date": r.get("date", ""),
|
| 241 |
+
}
|
| 242 |
+
if r.get("image"):
|
| 243 |
+
entry["image"] = r["image"]
|
| 244 |
+
results.append(entry)
|
| 245 |
+
|
| 246 |
+
elif backend == "images":
|
| 247 |
+
hits = list(ddgs.images(query, max_results=max_results, region=region))
|
| 248 |
+
results = []
|
| 249 |
+
for r in hits:
|
| 250 |
+
results.append({
|
| 251 |
+
"title": r.get("title", ""),
|
| 252 |
+
"url": r.get("image", ""),
|
| 253 |
+
"source": r.get("source", ""),
|
| 254 |
+
"thumbnail": r.get("thumbnail", ""),
|
| 255 |
+
"width": r.get("width"),
|
| 256 |
+
"height": r.get("height"),
|
| 257 |
+
})
|
| 258 |
+
|
| 259 |
+
else: # text
|
| 260 |
+
hits = list(ddgs.text(query, max_results=max_results, region=region))
|
| 261 |
+
results = []
|
| 262 |
+
for r in hits:
|
| 263 |
+
results.append({
|
| 264 |
+
"title": r.get("title", ""),
|
| 265 |
+
"url": r.get("href", ""),
|
| 266 |
+
"content": r.get("body", ""),
|
| 267 |
+
})
|
| 268 |
+
|
| 269 |
+
# Optionally extract full page content from top results
|
| 270 |
+
if extract_top > 0 and results and backend != "images":
|
| 271 |
+
urls_to_extract = [r["url"] for r in results[:extract_top] if r.get("url")]
|
| 272 |
+
with DDGS() as ddgs:
|
| 273 |
+
for url in urls_to_extract:
|
| 274 |
+
try:
|
| 275 |
+
extracted = ddgs.extract(url)
|
| 276 |
+
page_content = extracted.get("content", "")
|
| 277 |
+
if page_content:
|
| 278 |
+
# Find the matching result and add extracted content
|
| 279 |
+
for r in results:
|
| 280 |
+
if r["url"] == url:
|
| 281 |
+
r["extracted_content"] = page_content[:8000]
|
| 282 |
+
break
|
| 283 |
+
except Exception as e:
|
| 284 |
+
_log(f"Warning: failed to extract {url}: {e}")
|
| 285 |
+
|
| 286 |
return {
|
| 287 |
"query": query,
|
| 288 |
+
"backend": backend,
|
| 289 |
"results": results,
|
| 290 |
"total_results": len(results),
|
| 291 |
}
|
|
|
|
| 293 |
except ImportError:
|
| 294 |
_log("Warning: duckduckgo_search not installed, search disabled")
|
| 295 |
|
| 296 |
+
def _search_web(query, backend="text", max_results=10, extract_top=0, region="wt-wt"):
|
| 297 |
raise RuntimeError("Search not available: duckduckgo_search not installed")
|
| 298 |
|
| 299 |
|
|
|
|
| 398 |
return JSONResponse({**API_SPECS, "runtime": _get_runtime_specs()})
|
| 399 |
|
| 400 |
@fa_app.get("/respite/search")
|
| 401 |
+
def _search(q: str = "", backend: str = "text", max_results: int = 10, extract_top: int = 0, region: str = "wt-wt"):
|
| 402 |
if not q:
|
| 403 |
return JSONResponse({"error": "Missing query parameter 'q'"}, status_code=400)
|
| 404 |
+
if backend not in ("text", "news", "images"):
|
| 405 |
+
return JSONResponse({"error": f"Invalid backend '{backend}'. Use 'text', 'news', or 'images'."}, status_code=400)
|
| 406 |
try:
|
| 407 |
+
results = _search_web(q, backend=backend, max_results=max_results, extract_top=extract_top, region=region)
|
| 408 |
return JSONResponse(results)
|
| 409 |
except Exception as e:
|
| 410 |
return JSONResponse({"error": str(e)}, status_code=502)
|