mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-08-31 20:29:56 +02:00
[eric] web: move the search engines into tools/search so the folder stays under the item cap
This commit is contained in:
@@ -1,118 +0,0 @@
|
||||
"""DuckDuckGo web search: html endpoint primary, lite endpoint fallback.
|
||||
|
||||
The html endpoint is the richer parse; lite (see search_ddg_lite) covers the two
|
||||
ways html dies: the 202 bot challenge and silent markup drift. Only both
|
||||
endpoints challenging raises DDGRateLimited, so free search no longer has a
|
||||
single point of failure (the outage class that stranded subscription-only users
|
||||
on "No search backend is configured").
|
||||
|
||||
Both rungs go out through `browser_http`, whose Chrome TLS fingerprint is what
|
||||
actually decides whether DuckDuckGo answers; a plain httpx client scored 4/8 on
|
||||
the same queries this one scored 8/8 on."""
|
||||
|
||||
import html
|
||||
import re
|
||||
|
||||
from backend.apps.agents.tools.browser_http import CHROME_UA
|
||||
from backend.apps.agents.tools.browser_http import browser_request
|
||||
from backend.apps.agents.tools.search_ddg_lite import search_ddg_lite
|
||||
|
||||
HTTP_TIMEOUT = 30
|
||||
USER_AGENT = CHROME_UA
|
||||
|
||||
|
||||
class DDGRateLimited(Exception):
|
||||
"""Every DuckDuckGo frontend answered with the bot challenge (HTTP 202).
|
||||
|
||||
Named for history; this is an anti-automation challenge keyed on the
|
||||
client's fingerprint, NOT a per-IP rate limit. Distinct from 'genuinely
|
||||
zero hits' so the caller can fail over to another backend instead of
|
||||
reporting an empty search to the user."""
|
||||
|
||||
|
||||
def strip_html(raw_html: str) -> str:
|
||||
"""Naive but effective HTML to plain-text conversion."""
|
||||
text = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", raw_html, flags=re.DOTALL | re.IGNORECASE)
|
||||
text = re.sub(r"<[^>]+>", " ", text)
|
||||
text = html.unescape(text)
|
||||
text = re.sub(r"[ \t]+", " ", text)
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
async def search_ddg(query: str, num_results: int) -> str:
|
||||
"""Query DuckDuckGo's html endpoint and parse results; lite is the free fallback."""
|
||||
reply = await browser_request(
|
||||
"https://html.duckduckgo.com/html/", params={"q": query}, timeout=HTTP_TIMEOUT,
|
||||
)
|
||||
# DDG serves its bot challenge as 202 (a ~14KB no-results page), which is a 2xx so a status check sails right past it. Before giving up, try the lite frontend; only when BOTH challenge is free DDG actually dead.
|
||||
if reply.status == 202:
|
||||
lite = await search_ddg_lite(query, num_results)
|
||||
if lite is None:
|
||||
raise DDGRateLimited(query)
|
||||
return lite
|
||||
if reply.status >= 400:
|
||||
raise RuntimeError(f"DuckDuckGo html returned HTTP {reply.status}")
|
||||
|
||||
body = reply.text
|
||||
|
||||
result_blocks = re.findall(
|
||||
r'<div[^>]*class="[^"]*result[^"]*"[^>]*>(.*?)</div>\s*(?=<div[^>]*class="[^"]*result|$)',
|
||||
body,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
|
||||
entries: list[str] = []
|
||||
for block in result_blocks:
|
||||
if len(entries) >= num_results:
|
||||
break
|
||||
|
||||
# Handle both class-before-href and href-before-class attribute orders.
|
||||
link_match = re.search(
|
||||
r'<a[^>]*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)</a>',
|
||||
block,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
if not link_match:
|
||||
link_match = re.search(
|
||||
r'<a[^>]*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)</a>',
|
||||
block,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
if not link_match:
|
||||
continue
|
||||
|
||||
raw_url = html.unescape(link_match.group(1))
|
||||
|
||||
# Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results.
|
||||
if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url:
|
||||
continue
|
||||
|
||||
title = strip_html(link_match.group(2)).strip()
|
||||
|
||||
snippet_match = re.search(
|
||||
r'<a[^>]*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)</a>',
|
||||
block,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
snippet = strip_html(snippet_match.group(1)).strip() if snippet_match else ""
|
||||
|
||||
# DDG wraps URLs in a redirect; extract the real one.
|
||||
real_url_match = re.search(r"uddg=([^&]+)", raw_url)
|
||||
if real_url_match:
|
||||
from urllib.parse import unquote
|
||||
url = unquote(real_url_match.group(1))
|
||||
else:
|
||||
url = raw_url
|
||||
|
||||
entry = f"[{len(entries) + 1}] {title}\n {url}"
|
||||
if snippet:
|
||||
entry += f"\n {snippet}"
|
||||
entries.append(entry)
|
||||
|
||||
# 200 with zero parsed entries usually means DDG changed its markup out from under the regexes (it has before), not a genuine no-hits; lite's simpler shape is the safety net.
|
||||
if not entries:
|
||||
lite = await search_ddg_lite(query, num_results)
|
||||
if lite:
|
||||
return lite
|
||||
return "\n\n".join(entries)
|
||||
@@ -1,60 +0,0 @@
|
||||
"""DuckDuckGo lite-endpoint search: the fallback when html.duckduckgo.com
|
||||
serves its bot challenge (HTTP 202) or its markup drifts. lite.duckduckgo.com
|
||||
is a separate frontend with simpler, stabler HTML and direct result URLs (no
|
||||
uddg redirect).
|
||||
|
||||
Returns None on a challenge (caller decides whether that means every DDG
|
||||
frontend is closed) and a formatted results string (possibly empty) on
|
||||
success."""
|
||||
|
||||
import html
|
||||
import re
|
||||
from typing import List, Optional
|
||||
|
||||
from typeguard import typechecked
|
||||
|
||||
from backend.apps.agents.tools.browser_http import browser_request
|
||||
|
||||
P_LITE_URL = "https://lite.duckduckgo.com/lite/"
|
||||
P_TIMEOUT = 12.0
|
||||
P_TAG_RE = re.compile(r"<[^>]+>")
|
||||
# Lite uses single-quoted class attrs today; accept either quote style so a cosmetic flip doesn't kill the parser.
|
||||
P_LINK_RE = re.compile(
|
||||
r"""<a[^>]*href="([^"]+)"[^>]*class=['"]result-link['"][^>]*>(.*?)</a>""",
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
P_SNIPPET_RE = re.compile(
|
||||
r"""<td[^>]*class=['"]result-snippet['"][^>]*>(.*?)</td>""",
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
|
||||
|
||||
@typechecked
|
||||
def p_strip(text: str) -> str:
|
||||
return html.unescape(P_TAG_RE.sub("", text)).strip()
|
||||
|
||||
|
||||
@typechecked
|
||||
def parse_lite_results(body: str, num_results: int) -> str:
|
||||
"""Format lite's result rows; links and snippets appear in document order and pair up positionally."""
|
||||
links = P_LINK_RE.findall(body)
|
||||
snippets = [p_strip(s) for s in P_SNIPPET_RE.findall(body)]
|
||||
entries: List[str] = []
|
||||
for i, (url, raw_title) in enumerate(links[:num_results]):
|
||||
title = p_strip(raw_title)
|
||||
entry = f"[{i + 1}] {title}\n {html.unescape(url)}"
|
||||
if i < len(snippets) and snippets[i]:
|
||||
entry += f"\n {snippets[i]}"
|
||||
entries.append(entry)
|
||||
return "\n\n".join(entries)
|
||||
|
||||
|
||||
@typechecked
|
||||
async def search_ddg_lite(query: str, num_results: int) -> Optional[str]:
|
||||
"""None = bot challenge (202), string = parsed results (may be empty on no hits)."""
|
||||
reply = await browser_request(P_LITE_URL, params={"q": query}, timeout=P_TIMEOUT)
|
||||
if reply.status == 202:
|
||||
return None
|
||||
if reply.status >= 400:
|
||||
raise RuntimeError(f"DuckDuckGo lite returned HTTP {reply.status}")
|
||||
return parse_lite_results(reply.text, num_results)
|
||||
Reference in New Issue
Block a user