deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/community/ddg_search/tools.py
2026-09-07 18:24:55 +08:00

139 lines
4.2 KiB
Python

"""
Web Search Tool - Search using DuckDuckGo (no API key required).
DuckDuckGo's html endpoint rate-limits bursts of requests from the same IP
(403/429 challenge pages). The ``ddgs`` library raises ``DDGSException`` for
those (it does not retry), so rapid consecutive tool calls — e.g. a research
agent firing one query after another — fail on the 2nd/3rd call. We retry with
exponential backoff here and keep the error envelopes distinct so the calling
model can tell "no results" apart from "rate limited, retry later" instead of
inventing its own "network fluctuation" narrative.
"""
import json
import logging
import random
import time
from langchain.tools import tool
from deerflow.config import get_app_config
logger = logging.getLogger(__name__)
# Transient-failure retries (total attempts = 1 + _MAX_RETRIES).
_MAX_RETRIES = 2
_RETRY_DELAYS = (1.5, 3.0)
def _search_text(
query: str,
max_results: int = 5,
region: str = "wt-wt",
safesearch: str = "moderate",
) -> list[dict] | None:
"""
Execute text search using DuckDuckGo.
Args:
query: Search keywords
max_results: Maximum number of results
region: Search region
safesearch: Safe search level
Returns:
List of search results, or ``None`` when every attempt failed
(rate limited / network error) — the caller must not confuse that
with a genuine empty result set.
"""
try:
from ddgs import DDGS
except ImportError:
logger.error("ddgs library not installed. Run: pip install ddgs")
return None
ddgs = DDGS(timeout=30)
last_error: Exception | None = None
for attempt in range(_MAX_RETRIES + 1):
try:
results = ddgs.text(
query,
region=region,
safesearch=safesearch,
max_results=max_results,
)
return list(results) if results else []
except Exception as e:
last_error = e
if attempt < _MAX_RETRIES:
delay = _RETRY_DELAYS[attempt] + random.uniform(0, 0.5)
logger.warning(
"DuckDuckGo search attempt %s/%s failed (%s: %s); retrying in %.1fs",
attempt + 1,
_MAX_RETRIES + 1,
type(e).__name__,
e,
delay,
)
time.sleep(delay)
logger.error(
"Failed to search web after %s attempts: %s: %s",
_MAX_RETRIES + 1,
type(last_error).__name__ if last_error else "unknown",
last_error,
)
return None
@tool("web_search", parse_docstring=True)
def web_search_tool(
query: str,
max_results: int = 5,
) -> str:
"""Search the web for information. Use this tool to find current information, news, articles, and facts from the internet.
Args:
query: Search keywords describing what to find. Be specific for better results.
max_results: Maximum number of search results to return. Default is 5.
"""
config = get_app_config().get_tool_config("web_search")
# Override max_results from config if set
if config is not None and "max_results" in config.model_extra:
max_results = config.model_extra.get("max_results", max_results)
results = _search_text(
query=query,
max_results=max_results,
)
if results is None:
return json.dumps(
{
"error": ("web search failed after retries (rate limited or network error); wait briefly and retry once with different keywords instead of reporting a network problem to the user"),
"query": query,
},
ensure_ascii=False,
)
if not results:
return json.dumps({"error": "No results found", "query": query}, ensure_ascii=False)
normalized_results = [
{
"title": r.get("title", ""),
"url": r.get("href", r.get("link", "")),
"content": r.get("body", r.get("snippet", "")),
}
for r in results
]
output = {
"query": query,
"total_results": len(normalized_results),
"results": normalized_results,
}
return json.dumps(output, indent=2, ensure_ascii=False)