diff --git a/backend/apps/agents/tools/search_ddg.py b/backend/apps/agents/tools/search_ddg.py deleted file mode 100644 index 77084c2e..00000000 --- a/backend/apps/agents/tools/search_ddg.py +++ /dev/null @@ -1,118 +0,0 @@ -"""DuckDuckGo web search: html endpoint primary, lite endpoint fallback. - -The html endpoint is the richer parse; lite (see search_ddg_lite) covers the two -ways html dies: the 202 bot challenge and silent markup drift. Only both -endpoints challenging raises DDGRateLimited, so free search no longer has a -single point of failure (the outage class that stranded subscription-only users -on "No search backend is configured"). - -Both rungs go out through `browser_http`, whose Chrome TLS fingerprint is what -actually decides whether DuckDuckGo answers; a plain httpx client scored 4/8 on -the same queries this one scored 8/8 on.""" - -import html -import re - -from backend.apps.agents.tools.browser_http import CHROME_UA -from backend.apps.agents.tools.browser_http import browser_request -from backend.apps.agents.tools.search_ddg_lite import search_ddg_lite - -HTTP_TIMEOUT = 30 -USER_AGENT = CHROME_UA - - -class DDGRateLimited(Exception): - """Every DuckDuckGo frontend answered with the bot challenge (HTTP 202). - - Named for history; this is an anti-automation challenge keyed on the - client's fingerprint, NOT a per-IP rate limit. Distinct from 'genuinely - zero hits' so the caller can fail over to another backend instead of - reporting an empty search to the user.""" - - -def strip_html(raw_html: str) -> str: - """Naive but effective HTML to plain-text conversion.""" - text = re.sub(r"<(script|style)[^>]*>.*?", "", raw_html, flags=re.DOTALL | re.IGNORECASE) - text = re.sub(r"<[^>]+>", " ", text) - text = html.unescape(text) - text = re.sub(r"[ \t]+", " ", text) - text = re.sub(r"\n{3,}", "\n\n", text) - return text.strip() - - -async def search_ddg(query: str, num_results: int) -> str: - """Query DuckDuckGo's html endpoint and parse results; lite is the free fallback.""" - reply = await browser_request( - "https://html.duckduckgo.com/html/", params={"q": query}, timeout=HTTP_TIMEOUT, - ) - # DDG serves its bot challenge as 202 (a ~14KB no-results page), which is a 2xx so a status check sails right past it. Before giving up, try the lite frontend; only when BOTH challenge is free DDG actually dead. - if reply.status == 202: - lite = await search_ddg_lite(query, num_results) - if lite is None: - raise DDGRateLimited(query) - return lite - if reply.status >= 400: - raise RuntimeError(f"DuckDuckGo html returned HTTP {reply.status}") - - body = reply.text - - result_blocks = re.findall( - r']*class="[^"]*result[^"]*"[^>]*>(.*?)\s*(?=]*class="[^"]*result|$)', - body, - flags=re.DOTALL, - ) - - entries: list[str] = [] - for block in result_blocks: - if len(entries) >= num_results: - break - - # Handle both class-before-href and href-before-class attribute orders. - link_match = re.search( - r']*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)', - block, - flags=re.DOTALL, - ) - if not link_match: - link_match = re.search( - r']*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)', - block, - flags=re.DOTALL, - ) - if not link_match: - continue - - raw_url = html.unescape(link_match.group(1)) - - # Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results. - if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url: - continue - - title = strip_html(link_match.group(2)).strip() - - snippet_match = re.search( - r']*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)', - block, - flags=re.DOTALL, - ) - snippet = strip_html(snippet_match.group(1)).strip() if snippet_match else "" - - # DDG wraps URLs in a redirect; extract the real one. - real_url_match = re.search(r"uddg=([^&]+)", raw_url) - if real_url_match: - from urllib.parse import unquote - url = unquote(real_url_match.group(1)) - else: - url = raw_url - - entry = f"[{len(entries) + 1}] {title}\n {url}" - if snippet: - entry += f"\n {snippet}" - entries.append(entry) - - # 200 with zero parsed entries usually means DDG changed its markup out from under the regexes (it has before), not a genuine no-hits; lite's simpler shape is the safety net. - if not entries: - lite = await search_ddg_lite(query, num_results) - if lite: - return lite - return "\n\n".join(entries) diff --git a/backend/apps/agents/tools/search_ddg_lite.py b/backend/apps/agents/tools/search_ddg_lite.py deleted file mode 100644 index 77eae343..00000000 --- a/backend/apps/agents/tools/search_ddg_lite.py +++ /dev/null @@ -1,60 +0,0 @@ -"""DuckDuckGo lite-endpoint search: the fallback when html.duckduckgo.com -serves its bot challenge (HTTP 202) or its markup drifts. lite.duckduckgo.com -is a separate frontend with simpler, stabler HTML and direct result URLs (no -uddg redirect). - -Returns None on a challenge (caller decides whether that means every DDG -frontend is closed) and a formatted results string (possibly empty) on -success.""" - -import html -import re -from typing import List, Optional - -from typeguard import typechecked - -from backend.apps.agents.tools.browser_http import browser_request - -P_LITE_URL = "https://lite.duckduckgo.com/lite/" -P_TIMEOUT = 12.0 -P_TAG_RE = re.compile(r"<[^>]+>") -# Lite uses single-quoted class attrs today; accept either quote style so a cosmetic flip doesn't kill the parser. -P_LINK_RE = re.compile( - r"""]*href="([^"]+)"[^>]*class=['"]result-link['"][^>]*>(.*?)""", - flags=re.DOTALL, -) -P_SNIPPET_RE = re.compile( - r"""]*class=['"]result-snippet['"][^>]*>(.*?)""", - flags=re.DOTALL, -) - - -@typechecked -def p_strip(text: str) -> str: - return html.unescape(P_TAG_RE.sub("", text)).strip() - - -@typechecked -def parse_lite_results(body: str, num_results: int) -> str: - """Format lite's result rows; links and snippets appear in document order and pair up positionally.""" - links = P_LINK_RE.findall(body) - snippets = [p_strip(s) for s in P_SNIPPET_RE.findall(body)] - entries: List[str] = [] - for i, (url, raw_title) in enumerate(links[:num_results]): - title = p_strip(raw_title) - entry = f"[{i + 1}] {title}\n {html.unescape(url)}" - if i < len(snippets) and snippets[i]: - entry += f"\n {snippets[i]}" - entries.append(entry) - return "\n\n".join(entries) - - -@typechecked -async def search_ddg_lite(query: str, num_results: int) -> Optional[str]: - """None = bot challenge (202), string = parsed results (may be empty on no hits).""" - reply = await browser_request(P_LITE_URL, params={"q": query}, timeout=P_TIMEOUT) - if reply.status == 202: - return None - if reply.status >= 400: - raise RuntimeError(f"DuckDuckGo lite returned HTTP {reply.status}") - return parse_lite_results(reply.text, num_results)