[eric] web: DDG lite fallback so free search survives html-endpoint throttle/markup drift (+ split search_ddg out of web.py)

This commit is contained in:
ciregenz
2026-07-07 02:04:37 -07:00
parent 3ce848d185
commit 1db3022bd3
4 changed files with 305 additions and 100 deletions
+122
View File
@@ -0,0 +1,122 @@
"""DuckDuckGo web search: html endpoint primary, lite endpoint fallback.
The html endpoint is the richer parse; lite (see search_ddg_lite) covers the two
ways html dies: a 202 throttle and silent markup drift. Only both endpoints
throttling raises DDGRateLimited, so free search no longer has a single point
of failure (the outage class that stranded subscription-only users on
"No search backend is configured")."""
import html
import re
import httpx
from backend.apps.agents.tools.search_ddg_lite import search_ddg_lite
HTTP_TIMEOUT = 30
USER_AGENT = (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
class DDGRateLimited(Exception):
"""Both DuckDuckGo endpoints answered with the throttle challenge (HTTP 202).
Distinct from 'genuinely zero hits' so the caller can fail over to another
backend instead of reporting an empty search to the user. The throttle is
per-IP and burst-triggered; once BOTH html and lite serve it, the only cure
is a different backend or waiting it out."""
def strip_html(raw_html: str) -> str:
"""Naive but effective HTML to plain-text conversion."""
text = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", raw_html, flags=re.DOTALL | re.IGNORECASE)
text = re.sub(r"<[^>]+>", " ", text)
text = html.unescape(text)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
async def search_ddg(query: str, num_results: int) -> str:
"""Query DuckDuckGo's html endpoint and parse results; lite is the free fallback."""
async with httpx.AsyncClient(
timeout=HTTP_TIMEOUT,
follow_redirects=True,
headers={"User-Agent": USER_AGENT},
) as client:
resp = await client.post(
"https://html.duckduckgo.com/html/",
data={"q": query},
)
# DDG serves its throttle challenge as 202 (a ~14KB no-results page), which is a 2xx so raise_for_status() sails right past it. Before declaring rate-limited, try the lite frontend; only when BOTH throttle is free search actually dead.
if resp.status_code == 202:
lite = await search_ddg_lite(query, num_results)
if lite is None:
raise DDGRateLimited(query)
return lite
resp.raise_for_status()
body = resp.text
result_blocks = re.findall(
r'<div[^>]*class="[^"]*result[^"]*"[^>]*>(.*?)</div>\s*(?=<div[^>]*class="[^"]*result|$)',
body,
flags=re.DOTALL,
)
entries: list[str] = []
for block in result_blocks:
if len(entries) >= num_results:
break
# Handle both class-before-href and href-before-class attribute orders.
link_match = re.search(
r'<a[^>]*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)</a>',
block,
flags=re.DOTALL,
)
if not link_match:
link_match = re.search(
r'<a[^>]*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)</a>',
block,
flags=re.DOTALL,
)
if not link_match:
continue
raw_url = html.unescape(link_match.group(1))
# Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results.
if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url:
continue
title = strip_html(link_match.group(2)).strip()
snippet_match = re.search(
r'<a[^>]*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)</a>',
block,
flags=re.DOTALL,
)
snippet = strip_html(snippet_match.group(1)).strip() if snippet_match else ""
# DDG wraps URLs in a redirect; extract the real one.
real_url_match = re.search(r"uddg=([^&]+)", raw_url)
if real_url_match:
from urllib.parse import unquote
url = unquote(real_url_match.group(1))
else:
url = raw_url
entry = f"[{len(entries) + 1}] {title}\n {url}"
if snippet:
entry += f"\n {snippet}"
entries.append(entry)
# 200 with zero parsed entries usually means DDG changed its markup out from under the regexes (it has before), not a genuine no-hits; lite's simpler shape is the safety net.
if not entries:
lite = await search_ddg_lite(query, num_results)
if lite:
return lite
return "\n\n".join(entries)
@@ -0,0 +1,65 @@
"""DuckDuckGo lite-endpoint search: the free fallback when html.duckduckgo.com
throttles (HTTP 202) or its markup drifts. lite.duckduckgo.com is a separate
frontend with simpler, stabler HTML and direct result URLs (no uddg redirect).
Returns None on a throttle (caller decides whether that means rate-limited
overall) and a formatted results string (possibly empty) on success."""
import html
import re
from typing import List, Optional
import httpx
from typeguard import typechecked
P_LITE_URL = "https://lite.duckduckgo.com/lite/"
P_TIMEOUT = 12.0
P_USER_AGENT = (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
)
P_TAG_RE = re.compile(r"<[^>]+>")
# Lite uses single-quoted class attrs today; accept either quote style so a cosmetic flip doesn't kill the parser.
P_LINK_RE = re.compile(
r"""<a[^>]*href="([^"]+)"[^>]*class=['"]result-link['"][^>]*>(.*?)</a>""",
flags=re.DOTALL,
)
P_SNIPPET_RE = re.compile(
r"""<td[^>]*class=['"]result-snippet['"][^>]*>(.*?)</td>""",
flags=re.DOTALL,
)
@typechecked
def p_strip(text: str) -> str:
return html.unescape(P_TAG_RE.sub("", text)).strip()
@typechecked
def parse_lite_results(body: str, num_results: int) -> str:
"""Format lite's result rows; links and snippets appear in document order and pair up positionally."""
links = P_LINK_RE.findall(body)
snippets = [p_strip(s) for s in P_SNIPPET_RE.findall(body)]
entries: List[str] = []
for i, (url, raw_title) in enumerate(links[:num_results]):
title = p_strip(raw_title)
entry = f"[{i + 1}] {title}\n {html.unescape(url)}"
if i < len(snippets) and snippets[i]:
entry += f"\n {snippets[i]}"
entries.append(entry)
return "\n\n".join(entries)
@typechecked
async def search_ddg_lite(query: str, num_results: int) -> Optional[str]:
"""None = throttled (202), string = parsed results (may be empty on no hits)."""
async with httpx.AsyncClient(
timeout=P_TIMEOUT,
follow_redirects=True,
headers={"User-Agent": P_USER_AGENT},
) as client:
resp = await client.post(P_LITE_URL, data={"q": query})
if resp.status_code == 202:
return None
resp.raise_for_status()
return parse_lite_results(resp.text, num_results)
+11 -100
View File
@@ -2,7 +2,6 @@
from __future__ import annotations
import html
import re
from typing import Any, Optional
@@ -10,25 +9,18 @@ import httpx
from typeguard import typechecked
from backend.apps.agents.tools.base import BaseTool, ToolContext
from backend.apps.agents.tools.search_ddg import (
DDGRateLimited,
HTTP_TIMEOUT,
USER_AGENT,
strip_html,
)
from backend.apps.agents.tools.search_ddg import search_ddg as run_ddg_search
from backend.apps.agents.tools.ssrf_guard import SSRFBlocked, safe_fetch
P_HTTP_TIMEOUT = 30
P_MAX_OUTPUT_BYTES = 250 * 1024 # ~250 KB covers ~95% of articles/wikis/docs.
P_USER_AGENT = (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
)
class DDGRateLimited(Exception):
"""DuckDuckGo answered with its throttle challenge (HTTP 202), not results.
Distinct from 'genuinely zero hits' so the caller can fail over to another
backend instead of reporting an empty search to the user. The throttle is
per-IP and burst-triggered; a quick retry on the same or the `lite` endpoint
does NOT clear it (both share the limiter), so the only cure is a different
backend or waiting it out."""
def anthropic_web_search_is_reliable(*, uses_direct_anthropic_api: bool,
is_pro: bool) -> bool:
"""Whether the CLI's built-in WebSearch is reliable enough to suppress the
@@ -99,16 +91,6 @@ def p_truncate(text: str, limit: int = P_MAX_OUTPUT_BYTES) -> str:
return text
def p_strip_html(raw_html: str) -> str:
"""Naive but effective HTML to plain-text conversion."""
text = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", raw_html, flags=re.DOTALL | re.IGNORECASE)
text = re.sub(r"<[^>]+>", " ", text)
text = html.unescape(text)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
class WebSearchTool(BaseTool):
name = "WebSearch"
description = (
@@ -153,78 +135,7 @@ class WebSearchTool(BaseTool):
@staticmethod
async def search_ddg(query: str, num_results: int) -> str:
"""Query DuckDuckGo HTML endpoint and parse results."""
async with httpx.AsyncClient(
timeout=P_HTTP_TIMEOUT,
follow_redirects=True,
headers={"User-Agent": P_USER_AGENT},
) as client:
resp = await client.post(
"https://html.duckduckgo.com/html/",
data={"q": query},
)
# DDG serves its throttle challenge as 202 (a ~14KB no-results page), which is a 2xx so raise_for_status() sails right past it. Catch it explicitly so we report "rate-limited" instead of a bogus "no hits".
if resp.status_code == 202:
raise DDGRateLimited(query)
resp.raise_for_status()
body = resp.text
result_blocks = re.findall(
r'<div[^>]*class="[^"]*result[^"]*"[^>]*>(.*?)</div>\s*(?=<div[^>]*class="[^"]*result|$)',
body,
flags=re.DOTALL,
)
entries: list[str] = []
for block in result_blocks:
if len(entries) >= num_results:
break
# Handle both class-before-href and href-before-class attribute orders.
link_match = re.search(
r'<a[^>]*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)</a>',
block,
flags=re.DOTALL,
)
if not link_match:
link_match = re.search(
r'<a[^>]*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)</a>',
block,
flags=re.DOTALL,
)
if not link_match:
continue
raw_url = html.unescape(link_match.group(1))
# Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results.
if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url:
continue
title = p_strip_html(link_match.group(2)).strip()
snippet_match = re.search(
r'<a[^>]*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)</a>',
block,
flags=re.DOTALL,
)
snippet = p_strip_html(snippet_match.group(1)).strip() if snippet_match else ""
# DDG wraps URLs in a redirect; extract the real one.
real_url_match = re.search(r"uddg=([^&]+)", raw_url)
if real_url_match:
from urllib.parse import unquote
url = unquote(real_url_match.group(1))
else:
url = raw_url
entry = f"[{len(entries) + 1}] {title}\n {url}"
if snippet:
entry += f"\n {snippet}"
entries.append(entry)
return "\n\n".join(entries)
return await run_ddg_search(query, num_results)
class WebFetchTool(BaseTool):
@@ -259,8 +170,8 @@ class WebFetchTool(BaseTool):
resp = await safe_fetch(
url,
method="GET",
headers={"User-Agent": P_USER_AGENT},
timeout=P_HTTP_TIMEOUT,
headers={"User-Agent": USER_AGENT},
timeout=HTTP_TIMEOUT,
)
resp.raise_for_status()
except SSRFBlocked as exc:
@@ -287,7 +198,7 @@ class WebFetchTool(BaseTool):
except Exception:
text = None
if not text:
text = p_strip_html(resp.text)
text = strip_html(resp.text)
else:
text = resp.text
+107
View File
@@ -0,0 +1,107 @@
"""DDG lite fallback: free search must survive an html.duckduckgo.com throttle.
The field failure (Alex, 1.5.4): DDG's html endpoint 202-throttled and a
subscription-only user (no Gemini/OpenAI key) got "No results / no search
backend configured" for every query, free search had a single point of failure.
The seal: on html-202 OR html-parse-drift (200 with zero entries), search_ddg
falls to lite.duckduckgo.com; only BOTH throttling raises DDGRateLimited.
Network is mocked; the lite fixture is the real markup shape captured live
2026-07-07 (single-quoted class attrs, direct hrefs, paired snippet rows).
"""
import asyncio
import httpx
import pytest
from backend.apps.agents.tools.web import WebSearchTool, DDGRateLimited
from backend.apps.agents.tools.search_ddg_lite import parse_lite_results
P_LITE_BODY = """
<table>
<tr><td>1.&nbsp;</td><td>
<a rel="nofollow" href="https://www.anthropic.com/claude/fable" class='result-link'>Claude <b>Fable</b> \\ Anthropic</a>
</td></tr>
<tr><td>&nbsp;</td><td class='result-snippet'><b>Claude</b> <b>Fable</b> 5 is a real step forward.</td></tr>
<tr><td>2.&nbsp;</td><td>
<a rel="nofollow" href="https://techcrunch.com/2026/06/09/story" class='result-link'>TechCrunch story</a>
</td></tr>
<tr><td>&nbsp;</td><td class='result-snippet'>Second snippet text.</td></tr>
</table>
"""
P_HTML_202_BODY = "anomaly detected, challenge page"
class p_FakeResp:
def __init__(self, status_code: int, text: str):
self.status_code = status_code
self.text = text
def raise_for_status(self):
if self.status_code >= 400:
raise httpx.HTTPStatusError("err", request=None, response=None)
class p_RoutedClient:
"""Fake AsyncClient that answers per-URL, so the html and lite endpoints can behave differently in one test."""
def __init__(self, routes: dict):
self.routes = routes
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
async def post(self, url, *a, **k):
for key, resp in self.routes.items():
if key in url:
return resp
raise AssertionError(f"unexpected URL {url}")
def p_route(monkeypatch, routes: dict):
monkeypatch.setattr(httpx, "AsyncClient", lambda *a, **k: p_RoutedClient(routes))
def test_lite_parser_on_real_shape():
out = parse_lite_results(P_LITE_BODY, 5)
assert "[1] Claude Fable \\ Anthropic" in out
assert "https://www.anthropic.com/claude/fable" in out
assert "Claude Fable 5 is a real step forward." in out
assert "[2] TechCrunch story" in out
assert "Second snippet text." in out
def test_lite_parser_respects_num_results():
out = parse_lite_results(P_LITE_BODY, 1)
assert "[1]" in out and "[2]" not in out
def test_html_throttle_falls_to_lite(monkeypatch):
p_route(monkeypatch, {
"html.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY),
"lite.duckduckgo.com": p_FakeResp(200, P_LITE_BODY),
})
out = asyncio.run(WebSearchTool.search_ddg("q", 5))
assert "Claude Fable" in out # lite answered despite html throttle
def test_both_throttled_raises_rate_limited(monkeypatch):
p_route(monkeypatch, {
"html.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY),
"lite.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY),
})
with pytest.raises(DDGRateLimited):
asyncio.run(WebSearchTool.search_ddg("q", 5))
def test_html_markup_drift_falls_to_lite(monkeypatch):
p_route(monkeypatch, {
"html.duckduckgo.com": p_FakeResp(200, "<html><body>totally new markup</body></html>"),
"lite.duckduckgo.com": p_FakeResp(200, P_LITE_BODY),
})
out = asyncio.run(WebSearchTool.search_ddg("q", 5))
assert "Claude Fable" in out # drift didn't silently become "no results"