mirror of
https://github.com/openswarm-ai/openswarm.git
synced 2026-09-28 20:44:50 +02:00
[eric] web: DDG lite fallback so free search survives html-endpoint throttle/markup drift (+ split search_ddg out of web.py)
This commit is contained in:
@@ -0,0 +1,122 @@
|
||||
"""DuckDuckGo web search: html endpoint primary, lite endpoint fallback.
|
||||
|
||||
The html endpoint is the richer parse; lite (see search_ddg_lite) covers the two
|
||||
ways html dies: a 202 throttle and silent markup drift. Only both endpoints
|
||||
throttling raises DDGRateLimited, so free search no longer has a single point
|
||||
of failure (the outage class that stranded subscription-only users on
|
||||
"No search backend is configured")."""
|
||||
|
||||
import html
|
||||
import re
|
||||
|
||||
import httpx
|
||||
|
||||
from backend.apps.agents.tools.search_ddg_lite import search_ddg_lite
|
||||
|
||||
HTTP_TIMEOUT = 30
|
||||
USER_AGENT = (
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
|
||||
class DDGRateLimited(Exception):
|
||||
"""Both DuckDuckGo endpoints answered with the throttle challenge (HTTP 202).
|
||||
|
||||
Distinct from 'genuinely zero hits' so the caller can fail over to another
|
||||
backend instead of reporting an empty search to the user. The throttle is
|
||||
per-IP and burst-triggered; once BOTH html and lite serve it, the only cure
|
||||
is a different backend or waiting it out."""
|
||||
|
||||
|
||||
def strip_html(raw_html: str) -> str:
|
||||
"""Naive but effective HTML to plain-text conversion."""
|
||||
text = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", raw_html, flags=re.DOTALL | re.IGNORECASE)
|
||||
text = re.sub(r"<[^>]+>", " ", text)
|
||||
text = html.unescape(text)
|
||||
text = re.sub(r"[ \t]+", " ", text)
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
async def search_ddg(query: str, num_results: int) -> str:
|
||||
"""Query DuckDuckGo's html endpoint and parse results; lite is the free fallback."""
|
||||
async with httpx.AsyncClient(
|
||||
timeout=HTTP_TIMEOUT,
|
||||
follow_redirects=True,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
) as client:
|
||||
resp = await client.post(
|
||||
"https://html.duckduckgo.com/html/",
|
||||
data={"q": query},
|
||||
)
|
||||
# DDG serves its throttle challenge as 202 (a ~14KB no-results page), which is a 2xx so raise_for_status() sails right past it. Before declaring rate-limited, try the lite frontend; only when BOTH throttle is free search actually dead.
|
||||
if resp.status_code == 202:
|
||||
lite = await search_ddg_lite(query, num_results)
|
||||
if lite is None:
|
||||
raise DDGRateLimited(query)
|
||||
return lite
|
||||
resp.raise_for_status()
|
||||
|
||||
body = resp.text
|
||||
|
||||
result_blocks = re.findall(
|
||||
r'<div[^>]*class="[^"]*result[^"]*"[^>]*>(.*?)</div>\s*(?=<div[^>]*class="[^"]*result|$)',
|
||||
body,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
|
||||
entries: list[str] = []
|
||||
for block in result_blocks:
|
||||
if len(entries) >= num_results:
|
||||
break
|
||||
|
||||
# Handle both class-before-href and href-before-class attribute orders.
|
||||
link_match = re.search(
|
||||
r'<a[^>]*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)</a>',
|
||||
block,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
if not link_match:
|
||||
link_match = re.search(
|
||||
r'<a[^>]*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)</a>',
|
||||
block,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
if not link_match:
|
||||
continue
|
||||
|
||||
raw_url = html.unescape(link_match.group(1))
|
||||
|
||||
# Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results.
|
||||
if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url:
|
||||
continue
|
||||
|
||||
title = strip_html(link_match.group(2)).strip()
|
||||
|
||||
snippet_match = re.search(
|
||||
r'<a[^>]*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)</a>',
|
||||
block,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
snippet = strip_html(snippet_match.group(1)).strip() if snippet_match else ""
|
||||
|
||||
# DDG wraps URLs in a redirect; extract the real one.
|
||||
real_url_match = re.search(r"uddg=([^&]+)", raw_url)
|
||||
if real_url_match:
|
||||
from urllib.parse import unquote
|
||||
url = unquote(real_url_match.group(1))
|
||||
else:
|
||||
url = raw_url
|
||||
|
||||
entry = f"[{len(entries) + 1}] {title}\n {url}"
|
||||
if snippet:
|
||||
entry += f"\n {snippet}"
|
||||
entries.append(entry)
|
||||
|
||||
# 200 with zero parsed entries usually means DDG changed its markup out from under the regexes (it has before), not a genuine no-hits; lite's simpler shape is the safety net.
|
||||
if not entries:
|
||||
lite = await search_ddg_lite(query, num_results)
|
||||
if lite:
|
||||
return lite
|
||||
return "\n\n".join(entries)
|
||||
@@ -0,0 +1,65 @@
|
||||
"""DuckDuckGo lite-endpoint search: the free fallback when html.duckduckgo.com
|
||||
throttles (HTTP 202) or its markup drifts. lite.duckduckgo.com is a separate
|
||||
frontend with simpler, stabler HTML and direct result URLs (no uddg redirect).
|
||||
|
||||
Returns None on a throttle (caller decides whether that means rate-limited
|
||||
overall) and a formatted results string (possibly empty) on success."""
|
||||
|
||||
import html
|
||||
import re
|
||||
from typing import List, Optional
|
||||
|
||||
import httpx
|
||||
from typeguard import typechecked
|
||||
|
||||
P_LITE_URL = "https://lite.duckduckgo.com/lite/"
|
||||
P_TIMEOUT = 12.0
|
||||
P_USER_AGENT = (
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
||||
)
|
||||
P_TAG_RE = re.compile(r"<[^>]+>")
|
||||
# Lite uses single-quoted class attrs today; accept either quote style so a cosmetic flip doesn't kill the parser.
|
||||
P_LINK_RE = re.compile(
|
||||
r"""<a[^>]*href="([^"]+)"[^>]*class=['"]result-link['"][^>]*>(.*?)</a>""",
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
P_SNIPPET_RE = re.compile(
|
||||
r"""<td[^>]*class=['"]result-snippet['"][^>]*>(.*?)</td>""",
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
|
||||
|
||||
@typechecked
|
||||
def p_strip(text: str) -> str:
|
||||
return html.unescape(P_TAG_RE.sub("", text)).strip()
|
||||
|
||||
|
||||
@typechecked
|
||||
def parse_lite_results(body: str, num_results: int) -> str:
|
||||
"""Format lite's result rows; links and snippets appear in document order and pair up positionally."""
|
||||
links = P_LINK_RE.findall(body)
|
||||
snippets = [p_strip(s) for s in P_SNIPPET_RE.findall(body)]
|
||||
entries: List[str] = []
|
||||
for i, (url, raw_title) in enumerate(links[:num_results]):
|
||||
title = p_strip(raw_title)
|
||||
entry = f"[{i + 1}] {title}\n {html.unescape(url)}"
|
||||
if i < len(snippets) and snippets[i]:
|
||||
entry += f"\n {snippets[i]}"
|
||||
entries.append(entry)
|
||||
return "\n\n".join(entries)
|
||||
|
||||
|
||||
@typechecked
|
||||
async def search_ddg_lite(query: str, num_results: int) -> Optional[str]:
|
||||
"""None = throttled (202), string = parsed results (may be empty on no hits)."""
|
||||
async with httpx.AsyncClient(
|
||||
timeout=P_TIMEOUT,
|
||||
follow_redirects=True,
|
||||
headers={"User-Agent": P_USER_AGENT},
|
||||
) as client:
|
||||
resp = await client.post(P_LITE_URL, data={"q": query})
|
||||
if resp.status_code == 202:
|
||||
return None
|
||||
resp.raise_for_status()
|
||||
return parse_lite_results(resp.text, num_results)
|
||||
@@ -2,7 +2,6 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import re
|
||||
from typing import Any, Optional
|
||||
|
||||
@@ -10,25 +9,18 @@ import httpx
|
||||
from typeguard import typechecked
|
||||
|
||||
from backend.apps.agents.tools.base import BaseTool, ToolContext
|
||||
from backend.apps.agents.tools.search_ddg import (
|
||||
DDGRateLimited,
|
||||
HTTP_TIMEOUT,
|
||||
USER_AGENT,
|
||||
strip_html,
|
||||
)
|
||||
from backend.apps.agents.tools.search_ddg import search_ddg as run_ddg_search
|
||||
from backend.apps.agents.tools.ssrf_guard import SSRFBlocked, safe_fetch
|
||||
|
||||
P_HTTP_TIMEOUT = 30
|
||||
P_MAX_OUTPUT_BYTES = 250 * 1024 # ~250 KB covers ~95% of articles/wikis/docs.
|
||||
P_USER_AGENT = (
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
)
|
||||
|
||||
|
||||
class DDGRateLimited(Exception):
|
||||
"""DuckDuckGo answered with its throttle challenge (HTTP 202), not results.
|
||||
|
||||
Distinct from 'genuinely zero hits' so the caller can fail over to another
|
||||
backend instead of reporting an empty search to the user. The throttle is
|
||||
per-IP and burst-triggered; a quick retry on the same or the `lite` endpoint
|
||||
does NOT clear it (both share the limiter), so the only cure is a different
|
||||
backend or waiting it out."""
|
||||
|
||||
def anthropic_web_search_is_reliable(*, uses_direct_anthropic_api: bool,
|
||||
is_pro: bool) -> bool:
|
||||
"""Whether the CLI's built-in WebSearch is reliable enough to suppress the
|
||||
@@ -99,16 +91,6 @@ def p_truncate(text: str, limit: int = P_MAX_OUTPUT_BYTES) -> str:
|
||||
return text
|
||||
|
||||
|
||||
def p_strip_html(raw_html: str) -> str:
|
||||
"""Naive but effective HTML to plain-text conversion."""
|
||||
text = re.sub(r"<(script|style)[^>]*>.*?</\1>", "", raw_html, flags=re.DOTALL | re.IGNORECASE)
|
||||
text = re.sub(r"<[^>]+>", " ", text)
|
||||
text = html.unescape(text)
|
||||
text = re.sub(r"[ \t]+", " ", text)
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
class WebSearchTool(BaseTool):
|
||||
name = "WebSearch"
|
||||
description = (
|
||||
@@ -153,78 +135,7 @@ class WebSearchTool(BaseTool):
|
||||
|
||||
@staticmethod
|
||||
async def search_ddg(query: str, num_results: int) -> str:
|
||||
"""Query DuckDuckGo HTML endpoint and parse results."""
|
||||
async with httpx.AsyncClient(
|
||||
timeout=P_HTTP_TIMEOUT,
|
||||
follow_redirects=True,
|
||||
headers={"User-Agent": P_USER_AGENT},
|
||||
) as client:
|
||||
resp = await client.post(
|
||||
"https://html.duckduckgo.com/html/",
|
||||
data={"q": query},
|
||||
)
|
||||
# DDG serves its throttle challenge as 202 (a ~14KB no-results page), which is a 2xx so raise_for_status() sails right past it. Catch it explicitly so we report "rate-limited" instead of a bogus "no hits".
|
||||
if resp.status_code == 202:
|
||||
raise DDGRateLimited(query)
|
||||
resp.raise_for_status()
|
||||
|
||||
body = resp.text
|
||||
|
||||
result_blocks = re.findall(
|
||||
r'<div[^>]*class="[^"]*result[^"]*"[^>]*>(.*?)</div>\s*(?=<div[^>]*class="[^"]*result|$)',
|
||||
body,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
|
||||
entries: list[str] = []
|
||||
for block in result_blocks:
|
||||
if len(entries) >= num_results:
|
||||
break
|
||||
|
||||
# Handle both class-before-href and href-before-class attribute orders.
|
||||
link_match = re.search(
|
||||
r'<a[^>]*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)</a>',
|
||||
block,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
if not link_match:
|
||||
link_match = re.search(
|
||||
r'<a[^>]*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)</a>',
|
||||
block,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
if not link_match:
|
||||
continue
|
||||
|
||||
raw_url = html.unescape(link_match.group(1))
|
||||
|
||||
# Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results.
|
||||
if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url:
|
||||
continue
|
||||
|
||||
title = p_strip_html(link_match.group(2)).strip()
|
||||
|
||||
snippet_match = re.search(
|
||||
r'<a[^>]*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)</a>',
|
||||
block,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
snippet = p_strip_html(snippet_match.group(1)).strip() if snippet_match else ""
|
||||
|
||||
# DDG wraps URLs in a redirect; extract the real one.
|
||||
real_url_match = re.search(r"uddg=([^&]+)", raw_url)
|
||||
if real_url_match:
|
||||
from urllib.parse import unquote
|
||||
url = unquote(real_url_match.group(1))
|
||||
else:
|
||||
url = raw_url
|
||||
|
||||
entry = f"[{len(entries) + 1}] {title}\n {url}"
|
||||
if snippet:
|
||||
entry += f"\n {snippet}"
|
||||
entries.append(entry)
|
||||
|
||||
return "\n\n".join(entries)
|
||||
return await run_ddg_search(query, num_results)
|
||||
|
||||
|
||||
class WebFetchTool(BaseTool):
|
||||
@@ -259,8 +170,8 @@ class WebFetchTool(BaseTool):
|
||||
resp = await safe_fetch(
|
||||
url,
|
||||
method="GET",
|
||||
headers={"User-Agent": P_USER_AGENT},
|
||||
timeout=P_HTTP_TIMEOUT,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
timeout=HTTP_TIMEOUT,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
except SSRFBlocked as exc:
|
||||
@@ -287,7 +198,7 @@ class WebFetchTool(BaseTool):
|
||||
except Exception:
|
||||
text = None
|
||||
if not text:
|
||||
text = p_strip_html(resp.text)
|
||||
text = strip_html(resp.text)
|
||||
else:
|
||||
text = resp.text
|
||||
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
"""DDG lite fallback: free search must survive an html.duckduckgo.com throttle.
|
||||
|
||||
The field failure (Alex, 1.5.4): DDG's html endpoint 202-throttled and a
|
||||
subscription-only user (no Gemini/OpenAI key) got "No results / no search
|
||||
backend configured" for every query, free search had a single point of failure.
|
||||
The seal: on html-202 OR html-parse-drift (200 with zero entries), search_ddg
|
||||
falls to lite.duckduckgo.com; only BOTH throttling raises DDGRateLimited.
|
||||
|
||||
Network is mocked; the lite fixture is the real markup shape captured live
|
||||
2026-07-07 (single-quoted class attrs, direct hrefs, paired snippet rows).
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from backend.apps.agents.tools.web import WebSearchTool, DDGRateLimited
|
||||
from backend.apps.agents.tools.search_ddg_lite import parse_lite_results
|
||||
|
||||
P_LITE_BODY = """
|
||||
<table>
|
||||
<tr><td>1. </td><td>
|
||||
<a rel="nofollow" href="https://www.anthropic.com/claude/fable" class='result-link'>Claude <b>Fable</b> \\ Anthropic</a>
|
||||
</td></tr>
|
||||
<tr><td> </td><td class='result-snippet'><b>Claude</b> <b>Fable</b> 5 is a real step forward.</td></tr>
|
||||
<tr><td>2. </td><td>
|
||||
<a rel="nofollow" href="https://techcrunch.com/2026/06/09/story" class='result-link'>TechCrunch story</a>
|
||||
</td></tr>
|
||||
<tr><td> </td><td class='result-snippet'>Second snippet text.</td></tr>
|
||||
</table>
|
||||
"""
|
||||
|
||||
P_HTML_202_BODY = "anomaly detected, challenge page"
|
||||
|
||||
|
||||
class p_FakeResp:
|
||||
def __init__(self, status_code: int, text: str):
|
||||
self.status_code = status_code
|
||||
self.text = text
|
||||
|
||||
def raise_for_status(self):
|
||||
if self.status_code >= 400:
|
||||
raise httpx.HTTPStatusError("err", request=None, response=None)
|
||||
|
||||
|
||||
class p_RoutedClient:
|
||||
"""Fake AsyncClient that answers per-URL, so the html and lite endpoints can behave differently in one test."""
|
||||
def __init__(self, routes: dict):
|
||||
self.routes = routes
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
async def post(self, url, *a, **k):
|
||||
for key, resp in self.routes.items():
|
||||
if key in url:
|
||||
return resp
|
||||
raise AssertionError(f"unexpected URL {url}")
|
||||
|
||||
|
||||
def p_route(monkeypatch, routes: dict):
|
||||
monkeypatch.setattr(httpx, "AsyncClient", lambda *a, **k: p_RoutedClient(routes))
|
||||
|
||||
|
||||
def test_lite_parser_on_real_shape():
|
||||
out = parse_lite_results(P_LITE_BODY, 5)
|
||||
assert "[1] Claude Fable \\ Anthropic" in out
|
||||
assert "https://www.anthropic.com/claude/fable" in out
|
||||
assert "Claude Fable 5 is a real step forward." in out
|
||||
assert "[2] TechCrunch story" in out
|
||||
assert "Second snippet text." in out
|
||||
|
||||
|
||||
def test_lite_parser_respects_num_results():
|
||||
out = parse_lite_results(P_LITE_BODY, 1)
|
||||
assert "[1]" in out and "[2]" not in out
|
||||
|
||||
|
||||
def test_html_throttle_falls_to_lite(monkeypatch):
|
||||
p_route(monkeypatch, {
|
||||
"html.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY),
|
||||
"lite.duckduckgo.com": p_FakeResp(200, P_LITE_BODY),
|
||||
})
|
||||
out = asyncio.run(WebSearchTool.search_ddg("q", 5))
|
||||
assert "Claude Fable" in out # lite answered despite html throttle
|
||||
|
||||
|
||||
def test_both_throttled_raises_rate_limited(monkeypatch):
|
||||
p_route(monkeypatch, {
|
||||
"html.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY),
|
||||
"lite.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY),
|
||||
})
|
||||
with pytest.raises(DDGRateLimited):
|
||||
asyncio.run(WebSearchTool.search_ddg("q", 5))
|
||||
|
||||
|
||||
def test_html_markup_drift_falls_to_lite(monkeypatch):
|
||||
p_route(monkeypatch, {
|
||||
"html.duckduckgo.com": p_FakeResp(200, "<html><body>totally new markup</body></html>"),
|
||||
"lite.duckduckgo.com": p_FakeResp(200, P_LITE_BODY),
|
||||
})
|
||||
out = asyncio.run(WebSearchTool.search_ddg("q", 5))
|
||||
assert "Claude Fable" in out # drift didn't silently become "no results"
|
||||
Reference in New Issue
Block a user