From 1db3022bd3a4e75f8488b50b4799f6a1b289c275 Mon Sep 17 00:00:00 2001 From: ciregenz Date: Tue, 7 Jul 2026 02:04:37 -0700 Subject: [PATCH] [eric] web: DDG lite fallback so free search survives html-endpoint throttle/markup drift (+ split search_ddg out of web.py) --- backend/apps/agents/tools/search_ddg.py | 122 +++++++++++++++++++ backend/apps/agents/tools/search_ddg_lite.py | 65 ++++++++++ backend/apps/agents/tools/web.py | 111 ++--------------- backend/tests/test_web_search_ddg_lite.py | 107 ++++++++++++++++ 4 files changed, 305 insertions(+), 100 deletions(-) create mode 100644 backend/apps/agents/tools/search_ddg.py create mode 100644 backend/apps/agents/tools/search_ddg_lite.py create mode 100644 backend/tests/test_web_search_ddg_lite.py diff --git a/backend/apps/agents/tools/search_ddg.py b/backend/apps/agents/tools/search_ddg.py new file mode 100644 index 00000000..ba478251 --- /dev/null +++ b/backend/apps/agents/tools/search_ddg.py @@ -0,0 +1,122 @@ +"""DuckDuckGo web search: html endpoint primary, lite endpoint fallback. + +The html endpoint is the richer parse; lite (see search_ddg_lite) covers the two +ways html dies: a 202 throttle and silent markup drift. Only both endpoints +throttling raises DDGRateLimited, so free search no longer has a single point +of failure (the outage class that stranded subscription-only users on +"No search backend is configured").""" + +import html +import re + +import httpx + +from backend.apps.agents.tools.search_ddg_lite import search_ddg_lite + +HTTP_TIMEOUT = 30 +USER_AGENT = ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " + "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" +) + + +class DDGRateLimited(Exception): + """Both DuckDuckGo endpoints answered with the throttle challenge (HTTP 202). + + Distinct from 'genuinely zero hits' so the caller can fail over to another + backend instead of reporting an empty search to the user. The throttle is + per-IP and burst-triggered; once BOTH html and lite serve it, the only cure + is a different backend or waiting it out.""" + + +def strip_html(raw_html: str) -> str: + """Naive but effective HTML to plain-text conversion.""" + text = re.sub(r"<(script|style)[^>]*>.*?", "", raw_html, flags=re.DOTALL | re.IGNORECASE) + text = re.sub(r"<[^>]+>", " ", text) + text = html.unescape(text) + text = re.sub(r"[ \t]+", " ", text) + text = re.sub(r"\n{3,}", "\n\n", text) + return text.strip() + + +async def search_ddg(query: str, num_results: int) -> str: + """Query DuckDuckGo's html endpoint and parse results; lite is the free fallback.""" + async with httpx.AsyncClient( + timeout=HTTP_TIMEOUT, + follow_redirects=True, + headers={"User-Agent": USER_AGENT}, + ) as client: + resp = await client.post( + "https://html.duckduckgo.com/html/", + data={"q": query}, + ) + # DDG serves its throttle challenge as 202 (a ~14KB no-results page), which is a 2xx so raise_for_status() sails right past it. Before declaring rate-limited, try the lite frontend; only when BOTH throttle is free search actually dead. + if resp.status_code == 202: + lite = await search_ddg_lite(query, num_results) + if lite is None: + raise DDGRateLimited(query) + return lite + resp.raise_for_status() + + body = resp.text + + result_blocks = re.findall( + r']*class="[^"]*result[^"]*"[^>]*>(.*?)\s*(?=]*class="[^"]*result|$)', + body, + flags=re.DOTALL, + ) + + entries: list[str] = [] + for block in result_blocks: + if len(entries) >= num_results: + break + + # Handle both class-before-href and href-before-class attribute orders. + link_match = re.search( + r']*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)', + block, + flags=re.DOTALL, + ) + if not link_match: + link_match = re.search( + r']*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)', + block, + flags=re.DOTALL, + ) + if not link_match: + continue + + raw_url = html.unescape(link_match.group(1)) + + # Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results. + if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url: + continue + + title = strip_html(link_match.group(2)).strip() + + snippet_match = re.search( + r']*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)', + block, + flags=re.DOTALL, + ) + snippet = strip_html(snippet_match.group(1)).strip() if snippet_match else "" + + # DDG wraps URLs in a redirect; extract the real one. + real_url_match = re.search(r"uddg=([^&]+)", raw_url) + if real_url_match: + from urllib.parse import unquote + url = unquote(real_url_match.group(1)) + else: + url = raw_url + + entry = f"[{len(entries) + 1}] {title}\n {url}" + if snippet: + entry += f"\n {snippet}" + entries.append(entry) + + # 200 with zero parsed entries usually means DDG changed its markup out from under the regexes (it has before), not a genuine no-hits; lite's simpler shape is the safety net. + if not entries: + lite = await search_ddg_lite(query, num_results) + if lite: + return lite + return "\n\n".join(entries) diff --git a/backend/apps/agents/tools/search_ddg_lite.py b/backend/apps/agents/tools/search_ddg_lite.py new file mode 100644 index 00000000..fe5ac608 --- /dev/null +++ b/backend/apps/agents/tools/search_ddg_lite.py @@ -0,0 +1,65 @@ +"""DuckDuckGo lite-endpoint search: the free fallback when html.duckduckgo.com +throttles (HTTP 202) or its markup drifts. lite.duckduckgo.com is a separate +frontend with simpler, stabler HTML and direct result URLs (no uddg redirect). + +Returns None on a throttle (caller decides whether that means rate-limited +overall) and a formatted results string (possibly empty) on success.""" + +import html +import re +from typing import List, Optional + +import httpx +from typeguard import typechecked + +P_LITE_URL = "https://lite.duckduckgo.com/lite/" +P_TIMEOUT = 12.0 +P_USER_AGENT = ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36" +) +P_TAG_RE = re.compile(r"<[^>]+>") +# Lite uses single-quoted class attrs today; accept either quote style so a cosmetic flip doesn't kill the parser. +P_LINK_RE = re.compile( + r"""]*href="([^"]+)"[^>]*class=['"]result-link['"][^>]*>(.*?)""", + flags=re.DOTALL, +) +P_SNIPPET_RE = re.compile( + r"""]*class=['"]result-snippet['"][^>]*>(.*?)""", + flags=re.DOTALL, +) + + +@typechecked +def p_strip(text: str) -> str: + return html.unescape(P_TAG_RE.sub("", text)).strip() + + +@typechecked +def parse_lite_results(body: str, num_results: int) -> str: + """Format lite's result rows; links and snippets appear in document order and pair up positionally.""" + links = P_LINK_RE.findall(body) + snippets = [p_strip(s) for s in P_SNIPPET_RE.findall(body)] + entries: List[str] = [] + for i, (url, raw_title) in enumerate(links[:num_results]): + title = p_strip(raw_title) + entry = f"[{i + 1}] {title}\n {html.unescape(url)}" + if i < len(snippets) and snippets[i]: + entry += f"\n {snippets[i]}" + entries.append(entry) + return "\n\n".join(entries) + + +@typechecked +async def search_ddg_lite(query: str, num_results: int) -> Optional[str]: + """None = throttled (202), string = parsed results (may be empty on no hits).""" + async with httpx.AsyncClient( + timeout=P_TIMEOUT, + follow_redirects=True, + headers={"User-Agent": P_USER_AGENT}, + ) as client: + resp = await client.post(P_LITE_URL, data={"q": query}) + if resp.status_code == 202: + return None + resp.raise_for_status() + return parse_lite_results(resp.text, num_results) diff --git a/backend/apps/agents/tools/web.py b/backend/apps/agents/tools/web.py index 1ca140fc..ea637e99 100644 --- a/backend/apps/agents/tools/web.py +++ b/backend/apps/agents/tools/web.py @@ -2,7 +2,6 @@ from __future__ import annotations -import html import re from typing import Any, Optional @@ -10,25 +9,18 @@ import httpx from typeguard import typechecked from backend.apps.agents.tools.base import BaseTool, ToolContext +from backend.apps.agents.tools.search_ddg import ( + DDGRateLimited, + HTTP_TIMEOUT, + USER_AGENT, + strip_html, +) +from backend.apps.agents.tools.search_ddg import search_ddg as run_ddg_search from backend.apps.agents.tools.ssrf_guard import SSRFBlocked, safe_fetch -P_HTTP_TIMEOUT = 30 P_MAX_OUTPUT_BYTES = 250 * 1024 # ~250 KB covers ~95% of articles/wikis/docs. -P_USER_AGENT = ( - "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " - "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" -) -class DDGRateLimited(Exception): - """DuckDuckGo answered with its throttle challenge (HTTP 202), not results. - - Distinct from 'genuinely zero hits' so the caller can fail over to another - backend instead of reporting an empty search to the user. The throttle is - per-IP and burst-triggered; a quick retry on the same or the `lite` endpoint - does NOT clear it (both share the limiter), so the only cure is a different - backend or waiting it out.""" - def anthropic_web_search_is_reliable(*, uses_direct_anthropic_api: bool, is_pro: bool) -> bool: """Whether the CLI's built-in WebSearch is reliable enough to suppress the @@ -99,16 +91,6 @@ def p_truncate(text: str, limit: int = P_MAX_OUTPUT_BYTES) -> str: return text -def p_strip_html(raw_html: str) -> str: - """Naive but effective HTML to plain-text conversion.""" - text = re.sub(r"<(script|style)[^>]*>.*?", "", raw_html, flags=re.DOTALL | re.IGNORECASE) - text = re.sub(r"<[^>]+>", " ", text) - text = html.unescape(text) - text = re.sub(r"[ \t]+", " ", text) - text = re.sub(r"\n{3,}", "\n\n", text) - return text.strip() - - class WebSearchTool(BaseTool): name = "WebSearch" description = ( @@ -153,78 +135,7 @@ class WebSearchTool(BaseTool): @staticmethod async def search_ddg(query: str, num_results: int) -> str: - """Query DuckDuckGo HTML endpoint and parse results.""" - async with httpx.AsyncClient( - timeout=P_HTTP_TIMEOUT, - follow_redirects=True, - headers={"User-Agent": P_USER_AGENT}, - ) as client: - resp = await client.post( - "https://html.duckduckgo.com/html/", - data={"q": query}, - ) - # DDG serves its throttle challenge as 202 (a ~14KB no-results page), which is a 2xx so raise_for_status() sails right past it. Catch it explicitly so we report "rate-limited" instead of a bogus "no hits". - if resp.status_code == 202: - raise DDGRateLimited(query) - resp.raise_for_status() - - body = resp.text - - result_blocks = re.findall( - r']*class="[^"]*result[^"]*"[^>]*>(.*?)\s*(?=]*class="[^"]*result|$)', - body, - flags=re.DOTALL, - ) - - entries: list[str] = [] - for block in result_blocks: - if len(entries) >= num_results: - break - - # Handle both class-before-href and href-before-class attribute orders. - link_match = re.search( - r']*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)', - block, - flags=re.DOTALL, - ) - if not link_match: - link_match = re.search( - r']*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)', - block, - flags=re.DOTALL, - ) - if not link_match: - continue - - raw_url = html.unescape(link_match.group(1)) - - # Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results. - if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url: - continue - - title = p_strip_html(link_match.group(2)).strip() - - snippet_match = re.search( - r']*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)', - block, - flags=re.DOTALL, - ) - snippet = p_strip_html(snippet_match.group(1)).strip() if snippet_match else "" - - # DDG wraps URLs in a redirect; extract the real one. - real_url_match = re.search(r"uddg=([^&]+)", raw_url) - if real_url_match: - from urllib.parse import unquote - url = unquote(real_url_match.group(1)) - else: - url = raw_url - - entry = f"[{len(entries) + 1}] {title}\n {url}" - if snippet: - entry += f"\n {snippet}" - entries.append(entry) - - return "\n\n".join(entries) + return await run_ddg_search(query, num_results) class WebFetchTool(BaseTool): @@ -259,8 +170,8 @@ class WebFetchTool(BaseTool): resp = await safe_fetch( url, method="GET", - headers={"User-Agent": P_USER_AGENT}, - timeout=P_HTTP_TIMEOUT, + headers={"User-Agent": USER_AGENT}, + timeout=HTTP_TIMEOUT, ) resp.raise_for_status() except SSRFBlocked as exc: @@ -287,7 +198,7 @@ class WebFetchTool(BaseTool): except Exception: text = None if not text: - text = p_strip_html(resp.text) + text = strip_html(resp.text) else: text = resp.text diff --git a/backend/tests/test_web_search_ddg_lite.py b/backend/tests/test_web_search_ddg_lite.py new file mode 100644 index 00000000..41fc7485 --- /dev/null +++ b/backend/tests/test_web_search_ddg_lite.py @@ -0,0 +1,107 @@ +"""DDG lite fallback: free search must survive an html.duckduckgo.com throttle. + +The field failure (Alex, 1.5.4): DDG's html endpoint 202-throttled and a +subscription-only user (no Gemini/OpenAI key) got "No results / no search +backend configured" for every query, free search had a single point of failure. +The seal: on html-202 OR html-parse-drift (200 with zero entries), search_ddg +falls to lite.duckduckgo.com; only BOTH throttling raises DDGRateLimited. + +Network is mocked; the lite fixture is the real markup shape captured live +2026-07-07 (single-quoted class attrs, direct hrefs, paired snippet rows). +""" + +import asyncio + +import httpx +import pytest + +from backend.apps.agents.tools.web import WebSearchTool, DDGRateLimited +from backend.apps.agents.tools.search_ddg_lite import parse_lite_results + +P_LITE_BODY = """ + + + + + +
1.  + Claude Fable \\ Anthropic +
 Claude Fable 5 is a real step forward.
2.  + TechCrunch story +
 Second snippet text.
+""" + +P_HTML_202_BODY = "anomaly detected, challenge page" + + +class p_FakeResp: + def __init__(self, status_code: int, text: str): + self.status_code = status_code + self.text = text + + def raise_for_status(self): + if self.status_code >= 400: + raise httpx.HTTPStatusError("err", request=None, response=None) + + +class p_RoutedClient: + """Fake AsyncClient that answers per-URL, so the html and lite endpoints can behave differently in one test.""" + def __init__(self, routes: dict): + self.routes = routes + + async def __aenter__(self): + return self + + async def __aexit__(self, *a): + return False + + async def post(self, url, *a, **k): + for key, resp in self.routes.items(): + if key in url: + return resp + raise AssertionError(f"unexpected URL {url}") + + +def p_route(monkeypatch, routes: dict): + monkeypatch.setattr(httpx, "AsyncClient", lambda *a, **k: p_RoutedClient(routes)) + + +def test_lite_parser_on_real_shape(): + out = parse_lite_results(P_LITE_BODY, 5) + assert "[1] Claude Fable \\ Anthropic" in out + assert "https://www.anthropic.com/claude/fable" in out + assert "Claude Fable 5 is a real step forward." in out + assert "[2] TechCrunch story" in out + assert "Second snippet text." in out + + +def test_lite_parser_respects_num_results(): + out = parse_lite_results(P_LITE_BODY, 1) + assert "[1]" in out and "[2]" not in out + + +def test_html_throttle_falls_to_lite(monkeypatch): + p_route(monkeypatch, { + "html.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY), + "lite.duckduckgo.com": p_FakeResp(200, P_LITE_BODY), + }) + out = asyncio.run(WebSearchTool.search_ddg("q", 5)) + assert "Claude Fable" in out # lite answered despite html throttle + + +def test_both_throttled_raises_rate_limited(monkeypatch): + p_route(monkeypatch, { + "html.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY), + "lite.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY), + }) + with pytest.raises(DDGRateLimited): + asyncio.run(WebSearchTool.search_ddg("q", 5)) + + +def test_html_markup_drift_falls_to_lite(monkeypatch): + p_route(monkeypatch, { + "html.duckduckgo.com": p_FakeResp(200, "totally new markup"), + "lite.duckduckgo.com": p_FakeResp(200, P_LITE_BODY), + }) + out = asyncio.run(WebSearchTool.search_ddg("q", 5)) + assert "Claude Fable" in out # drift didn't silently become "no results"