diff --git a/backend/apps/agents/tools/search_ddg.py b/backend/apps/agents/tools/search_ddg.py
new file mode 100644
index 00000000..ba478251
--- /dev/null
+++ b/backend/apps/agents/tools/search_ddg.py
@@ -0,0 +1,122 @@
+"""DuckDuckGo web search: html endpoint primary, lite endpoint fallback.
+
+The html endpoint is the richer parse; lite (see search_ddg_lite) covers the two
+ways html dies: a 202 throttle and silent markup drift. Only both endpoints
+throttling raises DDGRateLimited, so free search no longer has a single point
+of failure (the outage class that stranded subscription-only users on
+"No search backend is configured")."""
+
+import html
+import re
+
+import httpx
+
+from backend.apps.agents.tools.search_ddg_lite import search_ddg_lite
+
+HTTP_TIMEOUT = 30
+USER_AGENT = (
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
+ "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
+)
+
+
+class DDGRateLimited(Exception):
+ """Both DuckDuckGo endpoints answered with the throttle challenge (HTTP 202).
+
+ Distinct from 'genuinely zero hits' so the caller can fail over to another
+ backend instead of reporting an empty search to the user. The throttle is
+ per-IP and burst-triggered; once BOTH html and lite serve it, the only cure
+ is a different backend or waiting it out."""
+
+
+def strip_html(raw_html: str) -> str:
+ """Naive but effective HTML to plain-text conversion."""
+ text = re.sub(r"<(script|style)[^>]*>.*?\1>", "", raw_html, flags=re.DOTALL | re.IGNORECASE)
+ text = re.sub(r"<[^>]+>", " ", text)
+ text = html.unescape(text)
+ text = re.sub(r"[ \t]+", " ", text)
+ text = re.sub(r"\n{3,}", "\n\n", text)
+ return text.strip()
+
+
+async def search_ddg(query: str, num_results: int) -> str:
+ """Query DuckDuckGo's html endpoint and parse results; lite is the free fallback."""
+ async with httpx.AsyncClient(
+ timeout=HTTP_TIMEOUT,
+ follow_redirects=True,
+ headers={"User-Agent": USER_AGENT},
+ ) as client:
+ resp = await client.post(
+ "https://html.duckduckgo.com/html/",
+ data={"q": query},
+ )
+ # DDG serves its throttle challenge as 202 (a ~14KB no-results page), which is a 2xx so raise_for_status() sails right past it. Before declaring rate-limited, try the lite frontend; only when BOTH throttle is free search actually dead.
+ if resp.status_code == 202:
+ lite = await search_ddg_lite(query, num_results)
+ if lite is None:
+ raise DDGRateLimited(query)
+ return lite
+ resp.raise_for_status()
+
+ body = resp.text
+
+ result_blocks = re.findall(
+ r'
]*class="[^"]*result|$)',
+ body,
+ flags=re.DOTALL,
+ )
+
+ entries: list[str] = []
+ for block in result_blocks:
+ if len(entries) >= num_results:
+ break
+
+ # Handle both class-before-href and href-before-class attribute orders.
+ link_match = re.search(
+ r'
]*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)',
+ block,
+ flags=re.DOTALL,
+ )
+ if not link_match:
+ link_match = re.search(
+ r'
]*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)',
+ block,
+ flags=re.DOTALL,
+ )
+ if not link_match:
+ continue
+
+ raw_url = html.unescape(link_match.group(1))
+
+ # Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results.
+ if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url:
+ continue
+
+ title = strip_html(link_match.group(2)).strip()
+
+ snippet_match = re.search(
+ r'
]*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)',
+ block,
+ flags=re.DOTALL,
+ )
+ snippet = strip_html(snippet_match.group(1)).strip() if snippet_match else ""
+
+ # DDG wraps URLs in a redirect; extract the real one.
+ real_url_match = re.search(r"uddg=([^&]+)", raw_url)
+ if real_url_match:
+ from urllib.parse import unquote
+ url = unquote(real_url_match.group(1))
+ else:
+ url = raw_url
+
+ entry = f"[{len(entries) + 1}] {title}\n {url}"
+ if snippet:
+ entry += f"\n {snippet}"
+ entries.append(entry)
+
+ # 200 with zero parsed entries usually means DDG changed its markup out from under the regexes (it has before), not a genuine no-hits; lite's simpler shape is the safety net.
+ if not entries:
+ lite = await search_ddg_lite(query, num_results)
+ if lite:
+ return lite
+ return "\n\n".join(entries)
diff --git a/backend/apps/agents/tools/search_ddg_lite.py b/backend/apps/agents/tools/search_ddg_lite.py
new file mode 100644
index 00000000..fe5ac608
--- /dev/null
+++ b/backend/apps/agents/tools/search_ddg_lite.py
@@ -0,0 +1,65 @@
+"""DuckDuckGo lite-endpoint search: the free fallback when html.duckduckgo.com
+throttles (HTTP 202) or its markup drifts. lite.duckduckgo.com is a separate
+frontend with simpler, stabler HTML and direct result URLs (no uddg redirect).
+
+Returns None on a throttle (caller decides whether that means rate-limited
+overall) and a formatted results string (possibly empty) on success."""
+
+import html
+import re
+from typing import List, Optional
+
+import httpx
+from typeguard import typechecked
+
+P_LITE_URL = "https://lite.duckduckgo.com/lite/"
+P_TIMEOUT = 12.0
+P_USER_AGENT = (
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
+ "(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
+)
+P_TAG_RE = re.compile(r"<[^>]+>")
+# Lite uses single-quoted class attrs today; accept either quote style so a cosmetic flip doesn't kill the parser.
+P_LINK_RE = re.compile(
+ r"""
]*href="([^"]+)"[^>]*class=['"]result-link['"][^>]*>(.*?)""",
+ flags=re.DOTALL,
+)
+P_SNIPPET_RE = re.compile(
+ r"""
]*class=['"]result-snippet['"][^>]*>(.*?) | """,
+ flags=re.DOTALL,
+)
+
+
+@typechecked
+def p_strip(text: str) -> str:
+ return html.unescape(P_TAG_RE.sub("", text)).strip()
+
+
+@typechecked
+def parse_lite_results(body: str, num_results: int) -> str:
+ """Format lite's result rows; links and snippets appear in document order and pair up positionally."""
+ links = P_LINK_RE.findall(body)
+ snippets = [p_strip(s) for s in P_SNIPPET_RE.findall(body)]
+ entries: List[str] = []
+ for i, (url, raw_title) in enumerate(links[:num_results]):
+ title = p_strip(raw_title)
+ entry = f"[{i + 1}] {title}\n {html.unescape(url)}"
+ if i < len(snippets) and snippets[i]:
+ entry += f"\n {snippets[i]}"
+ entries.append(entry)
+ return "\n\n".join(entries)
+
+
+@typechecked
+async def search_ddg_lite(query: str, num_results: int) -> Optional[str]:
+ """None = throttled (202), string = parsed results (may be empty on no hits)."""
+ async with httpx.AsyncClient(
+ timeout=P_TIMEOUT,
+ follow_redirects=True,
+ headers={"User-Agent": P_USER_AGENT},
+ ) as client:
+ resp = await client.post(P_LITE_URL, data={"q": query})
+ if resp.status_code == 202:
+ return None
+ resp.raise_for_status()
+ return parse_lite_results(resp.text, num_results)
diff --git a/backend/apps/agents/tools/web.py b/backend/apps/agents/tools/web.py
index 1ca140fc..ea637e99 100644
--- a/backend/apps/agents/tools/web.py
+++ b/backend/apps/agents/tools/web.py
@@ -2,7 +2,6 @@
from __future__ import annotations
-import html
import re
from typing import Any, Optional
@@ -10,25 +9,18 @@ import httpx
from typeguard import typechecked
from backend.apps.agents.tools.base import BaseTool, ToolContext
+from backend.apps.agents.tools.search_ddg import (
+ DDGRateLimited,
+ HTTP_TIMEOUT,
+ USER_AGENT,
+ strip_html,
+)
+from backend.apps.agents.tools.search_ddg import search_ddg as run_ddg_search
from backend.apps.agents.tools.ssrf_guard import SSRFBlocked, safe_fetch
-P_HTTP_TIMEOUT = 30
P_MAX_OUTPUT_BYTES = 250 * 1024 # ~250 KB covers ~95% of articles/wikis/docs.
-P_USER_AGENT = (
- "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
- "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
-)
-class DDGRateLimited(Exception):
- """DuckDuckGo answered with its throttle challenge (HTTP 202), not results.
-
- Distinct from 'genuinely zero hits' so the caller can fail over to another
- backend instead of reporting an empty search to the user. The throttle is
- per-IP and burst-triggered; a quick retry on the same or the `lite` endpoint
- does NOT clear it (both share the limiter), so the only cure is a different
- backend or waiting it out."""
-
def anthropic_web_search_is_reliable(*, uses_direct_anthropic_api: bool,
is_pro: bool) -> bool:
"""Whether the CLI's built-in WebSearch is reliable enough to suppress the
@@ -99,16 +91,6 @@ def p_truncate(text: str, limit: int = P_MAX_OUTPUT_BYTES) -> str:
return text
-def p_strip_html(raw_html: str) -> str:
- """Naive but effective HTML to plain-text conversion."""
- text = re.sub(r"<(script|style)[^>]*>.*?\1>", "", raw_html, flags=re.DOTALL | re.IGNORECASE)
- text = re.sub(r"<[^>]+>", " ", text)
- text = html.unescape(text)
- text = re.sub(r"[ \t]+", " ", text)
- text = re.sub(r"\n{3,}", "\n\n", text)
- return text.strip()
-
-
class WebSearchTool(BaseTool):
name = "WebSearch"
description = (
@@ -153,78 +135,7 @@ class WebSearchTool(BaseTool):
@staticmethod
async def search_ddg(query: str, num_results: int) -> str:
- """Query DuckDuckGo HTML endpoint and parse results."""
- async with httpx.AsyncClient(
- timeout=P_HTTP_TIMEOUT,
- follow_redirects=True,
- headers={"User-Agent": P_USER_AGENT},
- ) as client:
- resp = await client.post(
- "https://html.duckduckgo.com/html/",
- data={"q": query},
- )
- # DDG serves its throttle challenge as 202 (a ~14KB no-results page), which is a 2xx so raise_for_status() sails right past it. Catch it explicitly so we report "rate-limited" instead of a bogus "no hits".
- if resp.status_code == 202:
- raise DDGRateLimited(query)
- resp.raise_for_status()
-
- body = resp.text
-
- result_blocks = re.findall(
- r'
]*class="[^"]*result[^"]*"[^>]*>(.*?)
\s*(?=
]*class="[^"]*result|$)',
- body,
- flags=re.DOTALL,
- )
-
- entries: list[str] = []
- for block in result_blocks:
- if len(entries) >= num_results:
- break
-
- # Handle both class-before-href and href-before-class attribute orders.
- link_match = re.search(
- r'
]*class="[^"]*result__a[^"]*"[^>]*href="([^"]*)"[^>]*>(.*?)',
- block,
- flags=re.DOTALL,
- )
- if not link_match:
- link_match = re.search(
- r'
]*href="([^"]*)"[^>]*class="[^"]*result__a[^"]*"[^>]*>(.*?)',
- block,
- flags=re.DOTALL,
- )
- if not link_match:
- continue
-
- raw_url = html.unescape(link_match.group(1))
-
- # Drop sponsored rows: DDG ads point at its own y.js click-tracker (ad_domain/ad_provider) instead of a real uddg= redirect, so they'd otherwise show up as junk "duckduckgo.com/y.js?ad_..." results.
- if "/y.js?" in raw_url or "ad_provider=" in raw_url or "ad_domain=" in raw_url:
- continue
-
- title = p_strip_html(link_match.group(2)).strip()
-
- snippet_match = re.search(
- r'
]*class="[^"]*result__snippet[^"]*"[^>]*>(.*?)',
- block,
- flags=re.DOTALL,
- )
- snippet = p_strip_html(snippet_match.group(1)).strip() if snippet_match else ""
-
- # DDG wraps URLs in a redirect; extract the real one.
- real_url_match = re.search(r"uddg=([^&]+)", raw_url)
- if real_url_match:
- from urllib.parse import unquote
- url = unquote(real_url_match.group(1))
- else:
- url = raw_url
-
- entry = f"[{len(entries) + 1}] {title}\n {url}"
- if snippet:
- entry += f"\n {snippet}"
- entries.append(entry)
-
- return "\n\n".join(entries)
+ return await run_ddg_search(query, num_results)
class WebFetchTool(BaseTool):
@@ -259,8 +170,8 @@ class WebFetchTool(BaseTool):
resp = await safe_fetch(
url,
method="GET",
- headers={"User-Agent": P_USER_AGENT},
- timeout=P_HTTP_TIMEOUT,
+ headers={"User-Agent": USER_AGENT},
+ timeout=HTTP_TIMEOUT,
)
resp.raise_for_status()
except SSRFBlocked as exc:
@@ -287,7 +198,7 @@ class WebFetchTool(BaseTool):
except Exception:
text = None
if not text:
- text = p_strip_html(resp.text)
+ text = strip_html(resp.text)
else:
text = resp.text
diff --git a/backend/tests/test_web_search_ddg_lite.py b/backend/tests/test_web_search_ddg_lite.py
new file mode 100644
index 00000000..41fc7485
--- /dev/null
+++ b/backend/tests/test_web_search_ddg_lite.py
@@ -0,0 +1,107 @@
+"""DDG lite fallback: free search must survive an html.duckduckgo.com throttle.
+
+The field failure (Alex, 1.5.4): DDG's html endpoint 202-throttled and a
+subscription-only user (no Gemini/OpenAI key) got "No results / no search
+backend configured" for every query, free search had a single point of failure.
+The seal: on html-202 OR html-parse-drift (200 with zero entries), search_ddg
+falls to lite.duckduckgo.com; only BOTH throttling raises DDGRateLimited.
+
+Network is mocked; the lite fixture is the real markup shape captured live
+2026-07-07 (single-quoted class attrs, direct hrefs, paired snippet rows).
+"""
+
+import asyncio
+
+import httpx
+import pytest
+
+from backend.apps.agents.tools.web import WebSearchTool, DDGRateLimited
+from backend.apps.agents.tools.search_ddg_lite import parse_lite_results
+
+P_LITE_BODY = """
+
+"""
+
+P_HTML_202_BODY = "anomaly detected, challenge page"
+
+
+class p_FakeResp:
+ def __init__(self, status_code: int, text: str):
+ self.status_code = status_code
+ self.text = text
+
+ def raise_for_status(self):
+ if self.status_code >= 400:
+ raise httpx.HTTPStatusError("err", request=None, response=None)
+
+
+class p_RoutedClient:
+ """Fake AsyncClient that answers per-URL, so the html and lite endpoints can behave differently in one test."""
+ def __init__(self, routes: dict):
+ self.routes = routes
+
+ async def __aenter__(self):
+ return self
+
+ async def __aexit__(self, *a):
+ return False
+
+ async def post(self, url, *a, **k):
+ for key, resp in self.routes.items():
+ if key in url:
+ return resp
+ raise AssertionError(f"unexpected URL {url}")
+
+
+def p_route(monkeypatch, routes: dict):
+ monkeypatch.setattr(httpx, "AsyncClient", lambda *a, **k: p_RoutedClient(routes))
+
+
+def test_lite_parser_on_real_shape():
+ out = parse_lite_results(P_LITE_BODY, 5)
+ assert "[1] Claude Fable \\ Anthropic" in out
+ assert "https://www.anthropic.com/claude/fable" in out
+ assert "Claude Fable 5 is a real step forward." in out
+ assert "[2] TechCrunch story" in out
+ assert "Second snippet text." in out
+
+
+def test_lite_parser_respects_num_results():
+ out = parse_lite_results(P_LITE_BODY, 1)
+ assert "[1]" in out and "[2]" not in out
+
+
+def test_html_throttle_falls_to_lite(monkeypatch):
+ p_route(monkeypatch, {
+ "html.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY),
+ "lite.duckduckgo.com": p_FakeResp(200, P_LITE_BODY),
+ })
+ out = asyncio.run(WebSearchTool.search_ddg("q", 5))
+ assert "Claude Fable" in out # lite answered despite html throttle
+
+
+def test_both_throttled_raises_rate_limited(monkeypatch):
+ p_route(monkeypatch, {
+ "html.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY),
+ "lite.duckduckgo.com": p_FakeResp(202, P_HTML_202_BODY),
+ })
+ with pytest.raises(DDGRateLimited):
+ asyncio.run(WebSearchTool.search_ddg("q", 5))
+
+
+def test_html_markup_drift_falls_to_lite(monkeypatch):
+ p_route(monkeypatch, {
+ "html.duckduckgo.com": p_FakeResp(200, "totally new markup"),
+ "lite.duckduckgo.com": p_FakeResp(200, P_LITE_BODY),
+ })
+ out = asyncio.run(WebSearchTool.search_ddg("q", 5))
+ assert "Claude Fable" in out # drift didn't silently become "no results"