Web Search 1.0.0
web_search.py
5.2 KB · raw
"""Search the web, with no key and no account.
DuckDuckGo publishes no free search API, so this reads the plain HTML results
page a browser without JavaScript is given -- the same page, the same
results. It is unofficial: heavy use gets rate-limited, and a change to that
page breaks the parser here, which then says so rather than returning
nothing.
Only the standard library and httpx, which AWORG already depends on.
"""
from __future__ import annotations
import html
from html.parser import HTMLParser
from urllib.parse import parse_qs, unquote, urlparse
import httpx
from aworg.tools.base import ToolContext, ToolError, ToolResult
NAME = "web_search"
DESCRIPTION = (
"Search the web and get back titles, addresses and short snippets of the "
"top results. Read a result's page with a tool that fetches pages."
)
INPUT_SCHEMA = {
"type": "object",
"properties": {
"query": {"type": "string", "description": "What to search for."},
"count": {
"type": "integer",
"description": "How many results, 1 to 20. Defaults to 8.",
},
"region": {
"type": "string",
"description": (
"A DuckDuckGo region code to bias results, like 'us-en', "
"'uk-en' or 'de-de'. Defaults to no region."
),
},
},
"required": ["query"],
}
ENDPOINT = "https://html.duckduckgo.com/html/"
HEADERS = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/126.0 Safari/537.36"
),
"Accept-Language": "en-US,en;q=0.8",
}
class _Results(HTMLParser):
"""Pulls (title, href, snippet) out of the results page."""
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.results: list[dict[str, str]] = []
self._field: str | None = None
self._current: dict[str, str] | None = None
self.blocked = False
def handle_starttag(self, tag, attrs):
a = dict(attrs)
classes = (a.get("class") or "").split()
if "anomaly-modal" in " ".join(classes) or a.get("id") == "challenge-form":
self.blocked = True
if tag == "a" and "result__a" in classes:
self._current = {"title": "", "url": _real_url(a.get("href", "")), "snippet": ""}
self.results.append(self._current)
self._field = "title"
elif tag in ("a", "div") and "result__snippet" in classes and self._current is not None:
self._field = "snippet"
def handle_endtag(self, tag):
if tag in ("a", "div") and self._field:
self._field = None
def handle_data(self, data):
if self._field and self._current is not None:
self._current[self._field] += data
def _real_url(href: str) -> str:
"""Result links go through DuckDuckGo's redirect; take the target out."""
href = html.unescape(href)
if href.startswith("//"):
href = "https:" + href
parsed = urlparse(href)
if parsed.path.startswith("/l/"):
target = parse_qs(parsed.query).get("uddg", [""])[0]
if target:
return unquote(target)
return href
async def run(context: ToolContext, query: str = "", count: int = 8, region: str = "") -> ToolResult:
query = str(query).strip()
if not query:
raise ToolError("No query was given.")
try:
count = max(1, min(20, int(count or 8)))
except (TypeError, ValueError):
count = 8
form = {"q": query}
if region:
form["kl"] = str(region).strip()
try:
async with httpx.AsyncClient(headers=HEADERS, timeout=20, follow_redirects=True) as client:
response = await client.post(ENDPOINT, data=form)
except httpx.HTTPError as exc:
raise ToolError(f"Could not reach DuckDuckGo: {exc}") from None
if response.status_code in (202, 403, 429):
raise ToolError(
f"DuckDuckGo refused the search (HTTP {response.status_code}), which "
"it does when searches come too fast. Wait a minute and try again."
)
if response.status_code != 200:
raise ToolError(f"DuckDuckGo answered HTTP {response.status_code}.")
parser = _Results()
parser.feed(response.text)
if parser.blocked:
raise ToolError(
"DuckDuckGo asked for a human check instead of answering, which it "
"does when searches come too fast. Wait a minute and try again."
)
results = [
{k: " ".join(v.split()) for k, v in r.items()}
for r in parser.results
if r["url"].startswith("http") and "duckduckgo.com/y.js" not in r["url"]
][:count]
if not results:
if "result__a" not in response.text and "No results" not in response.text:
raise ToolError(
"DuckDuckGo's results page came back in a shape this tool does "
"not recognise, so nothing could be read from it."
)
return ToolResult(text=f"No results for {query!r}.", summary="no results")
lines = [f"{i}. {r['title']}\n {r['url']}\n {r['snippet']}" for i, r in enumerate(results, 1)]
return ToolResult(
text=f"Results for {query!r}:\n\n" + "\n\n".join(lines),
summary=f"{len(results)} results",
)