Merge pull request #2586 from laramies/fix/review-security-boundaries

fix: enforce discovery security boundaries
This commit is contained in:
Matt
2026-08-25 22:58:40 -04:00
committed by GitHub
29 changed files with 615 additions and 208 deletions
+1 -1
View File
@@ -67,7 +67,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- Completed bounded pagination for Wayback Archive and Cert Spotter, including continuation handling, truncation diagnostics, and preservation of partial results on provider failures ([df6ff2c9](https://github.com/laramies/theHarvester/commit/df6ff2c9), [f85a08ff](https://github.com/laramies/theHarvester/commit/f85a08ff)).
- Routed operator messages and diagnostics through logging, preserved host logging policy and existing handlers, and configured logging for the API service ([8a7b8b71](https://github.com/laramies/theHarvester/commit/8a7b8b71)).
- Added credential configuration adapters, deferred proxy configuration loading until required, and retained compatibility accessors such as `Core.brave_key()` ([ccd91176](https://github.com/laramies/theHarvester/commit/ccd91176)).
- Centralized hostname scope normalization for parser and storage boundaries, including case-insensitive targets, trailing dots, optional `www.` prefixes, and exact DNS-label matching ([c0a0b653](https://github.com/laramies/theHarvester/commit/c0a0b653), [a474f086](https://github.com/laramies/theHarvester/commit/a474f086), [70470cd8](https://github.com/laramies/theHarvester/commit/70470cd8)).
- Centralized hostname scope normalization for parser and storage boundaries, including case-insensitive targets, trailing dots, exact leading `www.` labels, and DNS-label boundary matching ([c0a0b653](https://github.com/laramies/theHarvester/commit/c0a0b653), [a474f086](https://github.com/laramies/theHarvester/commit/a474f086), [70470cd8](https://github.com/laramies/theHarvester/commit/70470cd8)).
- Replaced deprecated hostname resolution with `getaddrinfo`-based handling ([6a847435](https://github.com/laramies/theHarvester/commit/6a847435)).
- Reworked routine CI to use read-only permissions, non-mutating Ruff checks, offline tests, and explicit opt-in live provider checks ([72e5820f](https://github.com/laramies/theHarvester/commit/72e5820f)).
- Grouped GitHub Actions, Python, and Docker Dependabot updates with a seven-day cooldown, and added a seven-day `uv` dependency freshness window ([7a947b66](https://github.com/laramies/theHarvester/commit/7a947b66), [52a79cdb](https://github.com/laramies/theHarvester/commit/52a79cdb)).
+1
View File
@@ -23,6 +23,7 @@ When a change alters one of these boundaries, update this document and the neare
### Authorization and scope
- Every provider, DNS, or direct action stays within the operator's explicit target and selected activity. P0, P1, and P2 describe observable network behavior, not confidence or importance.
- The authorized hostname boundary is the operator's exact DNS name after canonicalization. A leading `www.` label is part of that boundary and is never stripped as a convenience alias of the registrable domain.
- P0 sources query existing providers or datasets. P1 actions query DNS about authorized names or addresses. P2 actions contact a target endpoint or cause equivalent direct interaction.
- Scope-extension candidates and external relationships remain review evidence. An operator decision is the only path that promotes them into a later run's authorized scope.
- ASN labels, registry records, BGP origins, RPKI states, DNS answers, and endpoint responses remain time-bound evidence. None establishes ownership, legal control, reachability, or authorization by itself.
+1 -1
View File
@@ -130,7 +130,7 @@ async def test_www_target_does_not_accept_sibling_subdomains(monkeypatch) -> Non
report = await search.process()
assert captured['params']['LOOKUP'] == 'example.com'
assert captured['params']['LOOKUP'] == 'www.example.com'
assert await search.get_hostnames() == {'www.example.com'}
assert await search.get_urls() == set()
assert report is None
+23 -2
View File
@@ -1,4 +1,6 @@
from argparse import Namespace
from contextlib import asynccontextmanager
from typing import Any
import pytest
@@ -43,6 +45,20 @@ class _Session:
return _Response(self.result.pop(0) if isinstance(self.result, list) else self.result)
@pytest.fixture(autouse=True)
def proxy_aware_session(monkeypatch: pytest.MonkeyPatch) -> list[dict[str, Any]]:
session_options: list[dict[str, Any]] = []
@asynccontextmanager
async def open_session(**kwargs: Any):
session_options.append(kwargs)
async with intelxsearch.aiohttp.ClientSession() as session:
yield session
monkeypatch.setattr(intelxsearch.AsyncFetcher, 'open_session', open_session)
return session_options
def test_blank_key_is_missing(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(intelxsearch.Core, 'intelx_key', staticmethod(lambda: ' '))
@@ -51,7 +67,10 @@ def test_blank_key_is_missing(monkeypatch: pytest.MonkeyPatch) -> None:
@pytest.mark.asyncio
async def test_process_exposes_flat_normalized_in_scope_results(monkeypatch: pytest.MonkeyPatch) -> None:
async def test_process_exposes_flat_normalized_in_scope_results(
monkeypatch: pytest.MonkeyPatch,
proxy_aware_session: list[dict[str, Any]],
) -> None:
result = {
'status': 1,
'selectors': [
@@ -80,11 +99,13 @@ async def test_process_exposes_flat_normalized_in_scope_results(monkeypatch: pyt
monkeypatch.setattr(intelxsearch.asyncio, 'sleep', no_sleep)
search = intelxsearch.SearchIntelx('example.com')
await search.process()
await search.process(proxy=True)
assert await search.get_emails() == ['admin@example.com']
assert await search.get_hostnames() == ['api.example.com', 'portal.example.com']
assert await search.get_urls() == ['https://portal.example.com/path']
assert len(proxy_aware_session) == 1
assert proxy_aware_session[0]['proxy'] is True
@pytest.mark.asyncio
+25 -4
View File
@@ -1,5 +1,7 @@
import sys
import types
from contextlib import asynccontextmanager
from typing import Any
import pytest
@@ -18,6 +20,19 @@ from theHarvester.discovery import rocketreach
from theHarvester.discovery.constants import MissingKey
@pytest.fixture(autouse=True)
def proxy_aware_session(monkeypatch: pytest.MonkeyPatch) -> list[dict[str, Any]]:
session_options: list[dict[str, Any]] = []
@asynccontextmanager
async def open_session(**kwargs: Any):
session_options.append(kwargs)
yield object()
monkeypatch.setattr(rocketreach.AsyncFetcher, 'open_session', open_session)
return session_options
@pytest.mark.asyncio
async def test_missing_key_raises(monkeypatch) -> None:
monkeypatch.setattr(rocketreach.Core, 'rocketreach_key', lambda: None)
@@ -26,7 +41,10 @@ async def test_missing_key_raises(monkeypatch) -> None:
@pytest.mark.asyncio
async def test_do_search_uses_people_data_endpoint_and_start_pagination(monkeypatch) -> None:
async def test_do_search_uses_people_data_endpoint_and_start_pagination(
monkeypatch: pytest.MonkeyPatch,
proxy_aware_session: list[dict[str, Any]],
) -> None:
monkeypatch.setattr(rocketreach.Core, 'rocketreach_key', lambda: 'test-key')
monkeypatch.setattr(rocketreach.Core, 'get_user_agent', lambda: 'test-agent')
monkeypatch.setattr(rocketreach, 'get_delay', lambda: 0)
@@ -70,11 +88,11 @@ async def test_do_search_uses_people_data_endpoint_and_start_pagination(monkeypa
monkeypatch.setattr(rocketreach.AsyncFetcher, 'post_fetch', fake_post_fetch)
search = rocketreach.SearchRocketReach('example.com', 150)
await search.process()
await search.process(proxy=True)
assert len(calls) == 2
first_url, first_headers, first_data, first_json, _ = calls[0]
second_url, _, second_data, _, _ = calls[1]
first_url, first_headers, first_data, first_json, first_kwargs = calls[0]
second_url, _, second_data, _, second_kwargs = calls[1]
assert first_url == 'https://api.rocketreach.co/api/v2/person/search'
assert second_url == 'https://api.rocketreach.co/api/v2/person/search'
@@ -83,6 +101,9 @@ async def test_do_search_uses_people_data_endpoint_and_start_pagination(monkeypa
assert first_json is True
assert first_data == {'query': {'current_employer_domain': ['example.com']}, 'start': 0, 'page_size': 100}
assert second_data == {'query': {'current_employer_domain': ['example.com']}, 'start': 100, 'page_size': 50}
assert first_kwargs['session'] is second_kwargs['session']
assert len(proxy_aware_session) == 1
assert proxy_aware_session[0]['proxy'] is True
links = await search.get_urls()
emails = await search.get_emails()
+11 -2
View File
@@ -248,7 +248,7 @@ class TestShodanEngine:
return FetcherResponse(
body={
'data': [{'ip_str': ip, 'port': 443, 'transport': 'tcp'}],
'hostnames': ['CDN.Example.TEST.', 'outside.test'],
'hostnames': ['CDN.WWW.Example.TEST.', 'outside.test'],
},
status=200,
headers={},
@@ -266,7 +266,7 @@ class TestShodanEngine:
'https://api.shodan.io/shodan/host/203.0.113.10',
'https://api.shodan.io/shodan/host/203.0.113.11',
]
assert await search.get_hostnames() == {'cdn.example.test'}
assert await search.get_hostnames() == {'cdn.www.example.test'}
assert report.status == 'failed'
assert report.stop_reason == 'provider-errors'
@@ -560,6 +560,15 @@ class TestShodanEngine:
assert report.status == 'failed'
assert report.stop_reason == 'dns-resolution-failed'
def test_shodan_preserves_an_explicit_www_search_scope(self, monkeypatch):
from theHarvester.discovery import shodansearch
monkeypatch.setattr(shodansearch.Core, 'shodan_key', lambda: 'test-key')
search = shodansearch.SearchShodan('WWW.Example.TEST.')
assert search.word == 'www.example.test'
@pytest.mark.asyncio
async def test_shodan_direct_request_cancellation_propagates(self, monkeypatch):
from theHarvester.discovery import shodansearch
+32 -1
View File
@@ -11,6 +11,7 @@ API documentation: https://ip.thc.org/docs/
from __future__ import annotations
from contextlib import asynccontextmanager
from typing import TYPE_CHECKING, Any, Self
from urllib.parse import parse_qs, urlparse
@@ -79,8 +80,18 @@ def session_for(*outcomes: FakeResponse | Exception) -> type[FakeSession]:
@pytest.fixture(autouse=True)
def fake_thc_session(monkeypatch: pytest.MonkeyPatch) -> None:
def fake_thc_session(monkeypatch: pytest.MonkeyPatch) -> list[dict[str, Any]]:
session_options: list[dict[str, Any]] = []
@asynccontextmanager
async def open_session(**kwargs: Any):
session_options.append(kwargs)
async with thc.aiohttp.ClientSession(**kwargs) as session:
yield session
monkeypatch.setattr(thc.aiohttp, 'ClientSession', FakeSession)
monkeypatch.setattr(thc.AsyncFetcher, 'open_session', open_session)
return session_options
@pytest.fixture
@@ -180,6 +191,26 @@ class TestThcSubdomainSearch:
result_list = list(result)
assert len(result_list) == len(set(result_list))
@pytest.mark.asyncio
async def test_proxy_is_stable_across_the_retry_conversation(
self,
monkeypatch: pytest.MonkeyPatch,
fake_thc_session: list[dict[str, Any]],
recorded_sleeps: list[int],
) -> None:
monkeypatch.setattr(
thc.aiohttp,
'ClientSession',
session_for(FakeResponse('', status=429), FakeResponse('api.example.com\n')),
)
report = await thc.SearchThc(self.domain()).process(proxy=True)
assert report is None
assert recorded_sleeps == [2]
assert len(fake_thc_session) == 1
assert fake_thc_session[0]['proxy'] is True
@pytest.mark.asyncio
async def test_unlimited_uses_provider_max_and_reports_saturation(
self,
+8
View File
@@ -88,6 +88,14 @@ def test_limit_must_be_a_positive_integer(monkeypatch: pytest.MonkeyPatch, limit
zoomeyesearch.SearchZoomEye('example.com', limit)
def test_explicit_www_target_is_not_widened(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(zoomeyesearch.Core, 'zoomeye_key', staticmethod(lambda: 'test-key'))
search = zoomeyesearch.SearchZoomEye('WWW.Example.COM.', 1)
assert search.target == 'www.example.com'
@pytest.mark.parametrize(
('response', 'status', 'reason'),
[
+73 -3
View File
@@ -886,6 +886,75 @@ def test_jsonl_normalizes_legacy_bare_asn_value() -> None:
assert findings == [{'type': 'asn', 'value': 'AS64500', 'sources': ['criminalip'], 'actions': []}]
@pytest.mark.parametrize(
('kind', 'value', 'expected'),
[
('hostname', 'API.Example.COM.', 'api.example.com'),
('ip', '192.0.2.010', None),
('ip', '2001:0DB8:0:0:0:0:0:1', '2001:db8::1'),
],
)
def test_jsonl_canonicalizes_hostname_and_ip_result_values(
kind: str,
value: str,
expected: str | None,
) -> None:
payload = '\n'.join(
(
json.dumps({'type': 'summary'}),
json.dumps({'type': kind, 'value': value, 'sources': [], 'actions': []}),
)
)
if expected is None:
with pytest.raises(ValueError, match='canonical IP'):
parse_result_jsonl(payload)
return
_summary, findings = parse_result_jsonl(payload)
assert findings[0]['value'] == expected
@pytest.mark.parametrize(
('kind', 'value', 'message'),
[
('hostname', 'bad host', 'canonical hostname'),
('ip', 'not-an-ip', 'canonical IP'),
],
)
def test_jsonl_rejects_invalid_hostname_and_ip_result_values(kind: str, value: str, message: str) -> None:
payload = '\n'.join(
(
json.dumps({'type': 'summary'}),
json.dumps({'type': kind, 'value': value, 'sources': [], 'actions': []}),
)
)
with pytest.raises(ValueError, match=message):
parse_result_jsonl(payload)
@pytest.mark.parametrize(
('kind', 'value', 'message'),
[
('hostname', 'bad host', 'hostname must be valid'),
('ip', 'not-an-ip', 'IP result must be a valid'),
('ip', 'fe80::1%eth0', 'IP result must not contain'),
],
)
def test_completed_result_rejects_invalid_hostname_and_ip_values(kind: str, value: str, message: str) -> None:
now = datetime(2026, 8, 25, 12, tzinfo=UTC)
with pytest.raises(ValueError, match=message):
CompletedResult.finish(
target='example.com',
started_at=now,
completed_at=now,
groups={kind: [value]},
)
@pytest.mark.parametrize(
'observation',
[
@@ -999,7 +1068,7 @@ def test_completed_result_rejects_structured_vhost_outside_the_run_scope(target:
)
def test_jsonl_rejects_noncanonical_structured_vhost_hostname() -> None:
def test_jsonl_canonicalizes_structured_vhost_hostname() -> None:
observation = vhost_observation('http://192.0.2.10', status=200, control_status=404)
details = {key: value for key, value in observation.to_record().items() if key not in {'type', 'hostname'}}
payload = '\n'.join(
@@ -1017,8 +1086,9 @@ def test_jsonl_rejects_noncanonical_structured_vhost_hostname() -> None:
)
)
with pytest.raises(ValueError, match='canonical hostname'):
parse_result_jsonl(payload)
_summary, findings = parse_result_jsonl(payload)
assert findings[0]['value'] == 'admin.example.com'
def test_jsonl_rejects_legacy_vhost_result_kind() -> None:
+77
View File
@@ -2,7 +2,10 @@ from __future__ import annotations
import asyncio
import logging
import stat
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
from threading import Barrier
from typing import Any
from unittest import mock
@@ -177,6 +180,38 @@ def test_read_config_copies_default_to_home(name: str, capsys):
assert got == expected
assert f"Created default {file.name} at {file}" in capsys.readouterr().out
assert file.exists()
assert stat.S_IMODE(file.parent.stat().st_mode) == 0o700
assert stat.S_IMODE(file.stat().st_mode) == 0o600
def test_read_config_concurrent_first_use_publishes_one_complete_file(monkeypatch: pytest.MonkeyPatch) -> None:
destination = CONFIG_DIRS[0].expanduser() / 'api-keys.yaml'
config_files = {directory.expanduser() / destination.name for directory in CONFIG_DIRS}
publication_barrier = Barrier(2)
link = core_module.os.link
read_text = Path.read_text
def isolated_read_text(path: Path, *args, **kwargs) -> str:
if path == destination and path.exists():
return read_text(path, *args, **kwargs)
if path in config_files:
raise FileNotFoundError
return read_text(path, *args, **kwargs)
def synchronized_link(source: str | Path, target: str | Path) -> None:
publication_barrier.wait(timeout=5)
link(source, target)
monkeypatch.setattr(core_module.os, 'link', synchronized_link)
monkeypatch.setattr(Path, 'read_text', isolated_read_text)
with ThreadPoolExecutor(max_workers=2) as executor:
configs = list(executor.map(lambda _index: Core._read_config('api-keys.yaml'), range(2)))
assert configs == [(DATA_DIR / 'api-keys.yaml').read_text()] * 2
assert destination.read_text() == configs[0]
assert stat.S_IMODE(destination.stat().st_mode) == 0o600
assert list(destination.parent.glob(f'.{destination.name}.*')) == []
_DEFAULT_JSON = object()
@@ -1168,6 +1203,48 @@ async def test_fetch_all_reuses_a_caller_owned_session(monkeypatch: pytest.Monke
assert seen_sessions == [session, session]
@pytest.mark.asyncio
async def test_fetch_all_rejects_proxy_selection_for_a_caller_owned_session() -> None:
with pytest.raises(ValueError, match='caller-owned session'):
await AsyncFetcher.fetch_all(['https://one.example'], proxy=True, session=object())
@pytest.mark.asyncio
async def test_fetch_all_resolves_one_socks_proxy_for_the_owned_session(monkeypatch: pytest.MonkeyPatch) -> None:
reset_dummy_sessions()
proxy_inputs: list[str | bool | None] = []
connector_inputs: list[tuple[str | None, str | None, object]] = []
monkeypatch.setattr(core_module.aiohttp, 'ClientSession', DummySession)
monkeypatch.setattr(AsyncFetcher, '_ssl_context', staticmethod(lambda _verify=True: 'ssl-context'))
def resolve_proxy(_cls: type[AsyncFetcher], proxy: str | bool | None) -> tuple[str | None, str | None]:
proxy_inputs.append(proxy)
return ('socks5://proxy.example:1080', 'socks5') if proxy else (None, None)
async def create_connector(
proxy_url: str | None,
proxy_type: str | None,
ssl_context: object,
) -> str:
connector_inputs.append((proxy_url, proxy_type, ssl_context))
return 'socks-connector'
monkeypatch.setattr(AsyncFetcher, '_resolve_proxy', classmethod(resolve_proxy))
monkeypatch.setattr(AsyncFetcher, '_create_connector', create_connector)
results = await AsyncFetcher.fetch_all(
['https://one.example', 'https://two.example'],
proxy=True,
)
assert results == ['response-text', 'response-text']
assert [proxy for proxy in proxy_inputs if proxy] == [True]
assert connector_inputs == [('socks5://proxy.example:1080', 'socks5', 'ssl-context')]
assert len(DummySession.instances) == 1
assert DummySession.instances[0].connector == 'socks-connector'
assert DummySession.instances[0].closed is True
@pytest.mark.asyncio
async def test_fetch_uses_http_proxy_when_enabled(monkeypatch) -> None:
reset_dummy_sessions()
+22
View File
@@ -0,0 +1,22 @@
from theHarvester.lib.hostnames import normalize_scoped_hostname
def test_normalize_scoped_hostname_keeps_www_as_the_boundary() -> None:
assert normalize_scoped_hostname('dev.www.example.com', 'WWW.Example.COM.') == 'dev.www.example.com'
assert normalize_scoped_hostname('www.example.com', 'WWW.Example.COM.') == 'www.example.com'
assert normalize_scoped_hostname('admin.example.com', 'WWW.Example.COM.') is None
def test_normalize_scoped_hostname_idna_encodes_value_and_target() -> None:
assert normalize_scoped_hostname('API.München.Example.TEST.', 'münchen.example.test') == 'api.xn--mnchen-3ya.example.test'
assert (
normalize_scoped_hostname('api.xn--mnchen-3ya.example.test', 'münchen.example.test') == 'api.xn--mnchen-3ya.example.test'
)
assert normalize_scoped_hostname('admin.example.test', 'münchen.example.test') is None
def test_normalize_scoped_hostname_rejects_invalid_or_unscoped_values() -> None:
assert normalize_scoped_hostname('bad_label.example.com', 'example.com') is None
assert normalize_scoped_hostname('192.0.2.1', 'example.com') is None
assert normalize_scoped_hostname('api.example.com', '192.0.2.1') is None
assert normalize_scoped_hostname(123, 'example.com') is None
+5 -1
View File
@@ -1,5 +1,5 @@
from theHarvester.discovery import apisguru, bevigil, builtwith, gitlabsearch, intelxsearch, rocketreach, urlscan, zoomeyesearch
from theHarvester.lib.source_catalog import SOURCE_SPECS, ActivityClass, ResultRoute, SourceSpec, get_source_spec
from theHarvester.lib.source_catalog import SOURCE_SPECS, ActivityClass, ResultRoute, SourceSpec, get_source_spec, resolve_sources
def test_dead_threatcrowd_source_is_not_selectable() -> None:
@@ -114,6 +114,10 @@ def test_source_lookup_preserves_case_insensitive_legacy_labels() -> None:
assert get_source_spec('CRTsh') is SOURCE_SPECS['crtsh']
def test_source_selection_canonicalizes_case_insensitive_legacy_labels() -> None:
assert resolve_sources('CRTsh,ShodanInternetDB') == ['crtsh', 'shodanInternetDB']
def test_crt_name_is_a_separate_passive_hostname_source() -> None:
spec = get_source_spec('CRT-NAME')
+71
View File
@@ -280,6 +280,77 @@ async def test_runner_normalizes_only_declared_apis_guru_routes(monkeypatch: pyt
assert outcome.asn_attributions == ()
@pytest.mark.asyncio
async def test_runner_preserves_an_explicit_www_target_boundary(monkeypatch: pytest.MonkeyPatch) -> None:
class FakeApisGuru:
async def process(self, _proxy: bool) -> None:
return None
async def get_hostnames(self) -> list[str]:
return ['www.example.com', 'dev.www.example.com', 'bad_label.www.example.com', 'admin.example.com']
async def get_emails(self) -> list[str]:
return []
async def get_urls(self) -> list[str]:
return []
monkeypatch.setitem(SOURCE_FACTORIES, 'apis-guru', lambda _request: FakeApisGuru())
outcome = await run_source(SourceRequest('apis-guru', 'www.example.com', 25, 0, False, True))
assert outcome.observations == (ResultObservation('apis-guru', 'hostname', 'dev.www.example.com'),)
@pytest.mark.asyncio
async def test_runner_keeps_idn_hostnames_inside_a_unicode_target(monkeypatch: pytest.MonkeyPatch) -> None:
class FakeApisGuru:
async def process(self, _proxy: bool) -> None:
return None
async def get_hostnames(self) -> list[str]:
return [
'münchen.example.test',
'api.münchen.example.test',
'api.xn--mnchen-3ya.example.test',
'admin.example.test',
]
async def get_emails(self) -> list[str]:
return []
async def get_urls(self) -> list[str]:
return []
monkeypatch.setitem(SOURCE_FACTORIES, 'apis-guru', lambda _request: FakeApisGuru())
outcome = await run_source(SourceRequest('apis-guru', 'münchen.example.test', 25, 0, False, True))
assert outcome.observations == (ResultObservation('apis-guru', 'hostname', 'api.xn--mnchen-3ya.example.test'),)
@pytest.mark.asyncio
async def test_runner_drops_ipv6_zone_identifiers(monkeypatch: pytest.MonkeyPatch) -> None:
class FakeOnyphe:
async def process(self, _proxy: bool) -> None:
return None
async def get_ips(self) -> list[str]:
return ['192.0.2.1', 'fe80::1%eth0', '2001:0db8:0:0:0:0:0:1']
async def get_asns(self) -> set[str]:
return set()
monkeypatch.setitem(SOURCE_FACTORIES, 'onyphe', lambda _request: FakeOnyphe())
outcome = await run_source(SourceRequest('onyphe', 'example.test', 25, 0, False, False))
assert outcome.observations == (
ResultObservation('onyphe', 'ip', '192.0.2.1'),
ResultObservation('onyphe', 'ip', '2001:db8::1'),
)
@pytest.mark.asyncio
async def test_runner_times_construction_and_records_missing_credentials(monkeypatch: pytest.MonkeyPatch) -> None:
ticks = iter((10.0, 10.125))
+44 -3
View File
@@ -558,17 +558,35 @@ async def test_virtual_host_output_keeps_legacy_lists_and_structured_jsonl(
assert finding['observations'][0]['endpoint'] == 'http://192.0.2.10:80/'
@pytest.mark.parametrize('target', ['Example.COM.', 'WWW.Example.COM.'])
def test_normalize_hosts_for_storage_uses_the_parser_scope(target: str) -> None:
@pytest.mark.parametrize(
('target', 'expected'),
[
('Example.COM.', {'api.example.com', 'dev.www.example.com', 'www.example.com'}),
('WWW.Example.COM.', {'dev.www.example.com'}),
('München.Example.TEST.', {'api.xn--mnchen-3ya.example.test'}),
],
)
def test_normalize_hosts_for_storage_uses_the_parser_scope(target: str, expected: set[str]) -> None:
discovered_hosts: set[object] = {
'API.Example.COM.',
'API.München.Example.TEST.',
'dev.www.example.com',
'www.example.com',
'example.com',
'badexample.com',
'bad_label.example.com',
'example.com.attacker.test',
123,
}
assert theharvester_main._normalize_hosts_for_storage(discovered_hosts, target) == {'api.example.com'}
assert theharvester_main._normalize_hosts_for_storage(discovered_hosts, target) == expected
def test_normalize_ip_addresses_drops_ipv6_zone_identifiers() -> None:
assert theharvester_main._normalize_ip_addresses(['192.0.2.1', 'fe80::1%eth0', '2001:0DB8:0:0:0:0:0:1', 123]) == {
'192.0.2.1',
'2001:db8::1',
}
@pytest.mark.asyncio
@@ -1523,6 +1541,29 @@ async def test_unlimited_subdomain_selection_passes_no_result_cap_to_every_sourc
assert all(job.request.limit is None for job in captured)
@pytest.mark.asyncio
async def test_cli_canonicalizes_a_case_insensitive_legacy_source(monkeypatch: pytest.MonkeyPatch) -> None:
captured: tuple[source_runner.SourceJob, ...] = ()
async def capture_jobs(
jobs: tuple[source_runner.SourceJob, ...],
**_kwargs: object,
) -> tuple[source_runner.SourceOutcome, ...]:
nonlocal captured
captured = jobs
return ()
monkeypatch.setattr(theharvester_main, 'run_source_jobs', capture_jobs)
monkeypatch.setattr(theharvester_main, 'ResultStore', _NoopResultStore)
await theharvester_main.start(
EnumerationOptions(domain='example.test', source='CRTsh', quiet=True),
return_completed_result=True,
)
assert [job.request.source for job in captured] == ['crtsh']
@pytest.mark.asyncio
async def test_source_completion_reports_verbose_terminal_summary(
monkeypatch: pytest.MonkeyPatch,
+20 -8
View File
@@ -8,18 +8,30 @@ from theHarvester.parsers import myparser
class TestMyParser(object):
@pytest.mark.asyncio
@pytest.mark.parametrize('word', ['Example.COM.', 'WWW.Example.COM.'])
async def test_emails_respect_target_label_boundaries(self, word: str) -> None:
results = "Admin@Example.COM***admin@notexample.com***.Lead@Sub.Example.COM.***other@outside.test"
@pytest.mark.parametrize(
('word', 'expected'),
[
('Example.COM.', {'admin@example.com', 'lead@sub.example.com', 'owner@www.example.com'}),
('WWW.Example.COM.', {'owner@www.example.com'}),
],
)
async def test_emails_respect_target_label_boundaries(self, word: str, expected: set[str]) -> None:
results = 'Admin@Example.COM***admin@notexample.com***.Lead@Sub.Example.COM.***owner@WWW.Example.COM***other@outside.test'
parse = myparser.Parser(results, word)
assert await parse.emails() == {"admin@example.com", "lead@sub.example.com"}
assert await parse.emails() == expected
@pytest.mark.asyncio
@pytest.mark.parametrize('word', ['Example.COM.', 'WWW.Example.COM.'])
async def test_hostnames_respect_target_label_boundaries(self, word: str) -> None:
results = "Example.COM. API.Example.COM. badexample.com outside.test sub.example.com"
@pytest.mark.parametrize(
('word', 'expected'),
[
('Example.COM.', {'api.example.com', 'example.com', 'sub.example.com', 'www.example.com'}),
('WWW.Example.COM.', {'www.example.com'}),
],
)
async def test_hostnames_respect_target_label_boundaries(self, word: str, expected: set[str]) -> None:
results = 'Example.COM. API.Example.COM. badexample.com outside.test sub.example.com WWW.Example.COM.'
parse = myparser.Parser(results, word)
assert set(await parse.hostnames()) == {"api.example.com", "example.com", "sub.example.com"}
assert set(await parse.hostnames()) == expected
@pytest.mark.asyncio
async def test_hostnames_remove_uppercase_encoded_slash(self) -> None:
+4 -4
View File
@@ -55,7 +55,7 @@ from theHarvester.lib.recursive_dns import (
discover_recursive_dns,
)
from theHarvester.lib.resolver_selection import DEFAULT_DNS_RESOLVERS, normalize_resolver_addresses
from theHarvester.lib.result_values import normalize_asn
from theHarvester.lib.result_values import normalize_asn, normalize_ip
from theHarvester.lib.routeviews import RouteViewsCancelled, RouteViewsResult, enrich_routeviews
from theHarvester.lib.shodan_evidence import ShodanHostObservation, canonical_shodan_hosts
from theHarvester.lib.source_catalog import (
@@ -91,11 +91,11 @@ logger = logging.getLogger(__name__)
def _normalize_hosts_for_storage(discovered_hosts: Iterable[object], target: str) -> set[str]:
normalized_target = target.strip().lower().removeprefix('www.').rstrip('.')
canonical_target = normalize_scoped_hostname(target, target)
return {
normalized
for host in discovered_hosts
if (normalized := normalize_scoped_hostname(host, normalized_target)) and normalized != normalized_target
if (normalized := normalize_scoped_hostname(host, target)) and normalized != canonical_target
}
@@ -105,7 +105,7 @@ def _normalize_ip_addresses(values: Iterable[object]) -> set[str]:
if not isinstance(value, str):
continue
try:
addresses.add(str(ip_address(value.strip())))
addresses.add(normalize_ip(value))
except ValueError:
continue
return addresses
+2 -6
View File
@@ -16,7 +16,6 @@ class SearchBuiltWith:
def __init__(self, word: str) -> None:
self.word = normalize_hostname(word)
self.lookup_domain = self.word.removeprefix('www.')
self.api_key = Core.builtwith_key()
if not isinstance(self.api_key, str) or not self.api_key.strip():
raise MissingKey('BuiltWith')
@@ -39,10 +38,7 @@ class SearchBuiltWith:
domain = normalize_hostname(domain)
except ValueError:
return None, True
normalized_domain = normalize_scoped_hostname(domain, self.lookup_domain)
if normalized_domain is None:
return None, False
candidate = normalized_domain if not subdomain.strip() else f'{subdomain.strip()}.{normalized_domain}'
candidate = domain if not subdomain.strip() else f'{subdomain.strip()}.{domain}'
try:
candidate = normalize_hostname(candidate)
except ValueError:
@@ -166,7 +162,7 @@ class SearchBuiltWith:
}
params = {
'HIDEDL': 'yes',
'LOOKUP': self.lookup_domain,
'LOOKUP': self.word,
'NOATTR': 'yes',
'NOMETA': 'yes',
'NOPII': 'yes',
+2 -2
View File
@@ -6,7 +6,7 @@ from urllib.parse import urlparse
import aiohttp
from theHarvester.discovery.constants import MissingKey
from theHarvester.lib.core import Core
from theHarvester.lib.core import AsyncFetcher, Core
from theHarvester.lib.hostnames import normalize_scoped_hostname
from theHarvester.lib.source_execution import SourceExecutionReport
from theHarvester.parsers import intelxparser
@@ -55,7 +55,7 @@ class SearchIntelx:
pending_polls = 0
try:
async with asyncio.timeout(self.MAX_RUNTIME_SECONDS):
async with aiohttp.ClientSession() as session:
async with AsyncFetcher.open_session(headers=headers, proxy=self.proxy) as session:
async with session.post(f'{self.database}/phonebook/search', headers=headers, json=data) as response:
if response.status in {401, 403}:
return SourceExecutionReport('failed', 'access-denied')
+49 -42
View File
@@ -34,53 +34,60 @@ class SearchRocketReach:
start = 0
remaining = self.limit
while remaining is None or remaining > 0:
page_size = min(100, remaining) if remaining is not None else 100
data = {
'query': {'current_employer_domain': [self.word]},
'start': start,
'page_size': page_size,
}
result = await AsyncFetcher.post_fetch(self.baseurl, headers=headers, data=data, json=True)
if not isinstance(result, dict):
break
detail = result.get('detail', '')
if detail and 'Subscribe to a plan to access' in str(detail):
# No more results can be fetched
break
if detail and 'Request was throttled.' in str(detail):
# Rate limit has been triggered need to sleep extra
logger.info(
f'RocketReach requests have been throttled; '
f'{str(detail).split(" ", 3)[-1].replace("available", "availability")}'
async with AsyncFetcher.open_session(headers=headers, proxy=self.proxy, request_timeout=720) as session:
while remaining is None or remaining > 0:
page_size = min(100, remaining) if remaining is not None else 100
data = {
'query': {'current_employer_domain': [self.word]},
'start': start,
'page_size': page_size,
}
result = await AsyncFetcher.post_fetch(
self.baseurl,
session=session,
headers=headers,
data=data,
json=True,
)
break
if not isinstance(result, dict):
break
profiles = result.get('profiles', [])
if not profiles:
break
detail = result.get('detail', '')
if detail and 'Subscribe to a plan to access' in str(detail):
# No more results can be fetched
break
for profile in profiles:
if 'linkedin_url' in profile:
self.urls.add(profile['linkedin_url'])
if profile.get('emails'):
for email in profile['emails']:
if email.get('email'):
self.emails.add(email['email'])
if detail and 'Request was throttled.' in str(detail):
# Rate limit has been triggered need to sleep extra
logger.info(
f'RocketReach requests have been throttled; '
f'{str(detail).split(" ", 3)[-1].replace("available", "availability")}'
)
break
found = len(profiles)
if remaining is not None:
remaining -= found
start += found
profiles = result.get('profiles', [])
if not profiles:
break
pagination = result.get('pagination', {})
total = pagination.get('total')
if isinstance(total, int) and start >= total:
break
if found < page_size:
break
for profile in profiles:
if 'linkedin_url' in profile:
self.urls.add(profile['linkedin_url'])
if profile.get('emails'):
for email in profile['emails']:
if email.get('email'):
self.emails.add(email['email'])
found = len(profiles)
if remaining is not None:
remaining -= found
start += found
pagination = result.get('pagination', {})
total = pagination.get('total')
if isinstance(total, int) and start >= total:
break
if found < page_size:
break
await asyncio.sleep(get_delay() + 5)
+22 -17
View File
@@ -14,7 +14,7 @@ from theHarvester.discovery.constants import MissingKey
from theHarvester.lib.asn_attribution import AsnAttributionObservation
from theHarvester.lib.core import AsyncFetcher, Core, FetcherResponse, ResponseStreamError
from theHarvester.lib.hostchecker import resolve_ip_addresses
from theHarvester.lib.hostnames import normalize_scoped_hostname
from theHarvester.lib.hostnames import normalize_hostname, normalize_scoped_hostname
from theHarvester.lib.shodan_evidence import ShodanHostObservation, canonical_shodan_hosts
from theHarvester.lib.source_execution import SourceExecutionReport, SourceReportStatus
@@ -67,10 +67,15 @@ class SearchShodan:
REQUEST_TIMEOUT_SECONDS: int | None = None
def __init__(self, word: str | None = None) -> None:
self.word = word.strip().lower().rstrip('.') if word is not None else None
if self.word is not None and (self.word.startswith('*.') or _CERTIFICATE_HOSTNAME.fullmatch(self.word) is None):
raise ValueError('Shodan discovery target must be a hostname')
self.scope = self.word.removeprefix('www.') if self.word is not None else None
if word is None:
self.word = None
else:
try:
self.word = normalize_hostname(word)
except ValueError as error:
raise ValueError('Shodan discovery target must be a hostname') from error
if _CERTIFICATE_HOSTNAME.fullmatch(self.word) is None:
raise ValueError('Shodan discovery target must be a hostname')
self.key = Core.shodan_key()
if self.key is None:
raise MissingKey('Shodan')
@@ -104,8 +109,8 @@ class SearchShodan:
return None
wildcard = candidate.startswith('*.')
hostname = candidate.removeprefix('*.')
if self.scope is not None:
hostname = normalize_scoped_hostname(hostname, self.scope) or ''
if self.word is not None:
hostname = normalize_scoped_hostname(hostname, self.word) or ''
if not hostname:
return None
else:
@@ -150,13 +155,13 @@ class SearchShodan:
return names, invalid_response
def _scoped_banner_names(self, banner: dict[str, object]) -> set[str]:
assert self.scope is not None
assert self.word is not None
names: set[str] = set()
for field in ('hostnames', 'domains'):
values = banner.get(field)
if isinstance(values, list):
for value in values:
if normalized := normalize_scoped_hostname(value, self.scope):
if normalized := normalize_scoped_hostname(value, self.word):
names.add(normalized)
ssl = banner.get('ssl')
if isinstance(ssl, dict):
@@ -323,14 +328,14 @@ class SearchShodan:
domain_values = normalized_strings(results.get('domains'))
hostname_values = normalized_strings(results.get('hostnames'))
if self.scope is not None:
if self.word is not None:
domain_values = sorted(
{normalized for value in domain_values if (normalized := normalize_scoped_hostname(value, self.scope))}
{normalized for value in domain_values if (normalized := normalize_scoped_hostname(value, self.word))}
)
hostname_values = sorted(
{normalized for value in hostname_values if (normalized := normalize_scoped_hostname(value, self.scope))}
{normalized for value in hostname_values if (normalized := normalize_scoped_hostname(value, self.word))}
)
self.totalhosts.update(value for value in domain_values + hostname_values if value != self.scope)
self.totalhosts.update(value for value in domain_values + hostname_values if value != self.word)
for service in services:
tls = service.get('tls')
if not isinstance(tls, dict):
@@ -339,7 +344,7 @@ class SearchShodan:
tls_value = tls.get(field)
values = tls_value if isinstance(tls_value, list) else [tls_value]
self.totalhosts.update(
name for name in values if isinstance(name, str) and not name.startswith('*.') and name != self.scope
name for name in values if isinstance(name, str) and not name.startswith('*.') and name != self.word
)
asn = normalized_string(results.get('asn'))
organization = normalized_string(results.get('org'))
@@ -371,9 +376,9 @@ class SearchShodan:
return invalid_response
async def _search_target(self, proxy: bool) -> set[str]:
assert self.scope is not None
assert self.word is not None
error_types: set[str] = set()
for query in (f'hostname:{self.scope}', f'ssl:{self.scope}'):
for query in (f'hostname:{self.word}', f'ssl:{self.word}'):
page = 1
received = 0
seen_pages: set[str] = set()
@@ -497,7 +502,7 @@ class SearchShodan:
async def process(self, proxy: bool = False) -> SourceExecutionReport | None:
if self.word is None:
raise ValueError('A discovery target is required')
assert self.scope is not None
assert self.word is not None
self.totalhosts.clear()
dns_stop_reason: str | None = None
+42 -37
View File
@@ -4,7 +4,7 @@ from urllib.parse import urlencode
import aiohttp
from theHarvester.lib.core import Core
from theHarvester.lib.core import AsyncFetcher, Core
from theHarvester.lib.hostnames import normalize_scoped_hostname
from theHarvester.lib.source_execution import SourceExecutionReport
@@ -32,46 +32,51 @@ class SearchThc:
url = f'https://ip.thc.org/api/v1/subdomains/download?{query}'
headers = {'User-Agent': Core.get_user_agent()}
for attempt in range(self.max_retries):
try:
timeout = aiohttp.ClientTimeout(total=60)
async with aiohttp.ClientSession(headers=headers, timeout=timeout) as session:
async with session.get(url) as response:
if response.status == 429:
rate_remaining = response.headers.get('x-ratelimit-remaining', '0')
try:
async with AsyncFetcher.open_session(headers=headers, proxy=self.proxy, request_timeout=60) as session:
for attempt in range(self.max_retries):
try:
async with session.get(url) as response:
if response.status == 429:
rate_remaining = response.headers.get('x-ratelimit-remaining', '0')
if attempt == self.max_retries - 1:
logger.info(f'THC returned status 429 after {self.max_retries} attempts')
return SourceExecutionReport('rate-limited', 'http-429')
wait_time = self.base_delay * (attempt + 1)
logger.info(
f'THC rate limit hit (remaining: {rate_remaining}). Waiting {wait_time}s before retry...'
)
await asyncio.sleep(wait_time)
continue
if response.status != 200:
logger.info(f'THC returned status {response.status}')
return SourceExecutionReport('failed', f'http-{response.status}')
text = await response.text()
lines = text.splitlines()
for line in lines:
if hostname := normalize_scoped_hostname(line, self.word):
self.results.add(hostname)
if len(lines) >= requested and (self.limit is None or self.limit > self.PROVIDER_MAX_RESULTS):
return SourceExecutionReport('partial', 'provider-limit')
return None
except Exception as e:
error_msg = str(e).lower()
if '429' in error_msg or 'rate' in error_msg:
if attempt == self.max_retries - 1:
logger.info(f'THC returned status 429 after {self.max_retries} attempts')
return SourceExecutionReport('rate-limited', 'http-429')
logger.info(f'THC rate limit failure after {self.max_retries} attempts')
return SourceExecutionReport('rate-limited', 'provider-rate-limit')
wait_time = self.base_delay * (attempt + 1)
logger.info(f'THC rate limit hit (remaining: {rate_remaining}). Waiting {wait_time}s before retry...')
logger.info(f'THC rate limit detected. Waiting {wait_time}s before retry...')
await asyncio.sleep(wait_time)
continue
if response.status != 200:
logger.info(f'THC returned status {response.status}')
return SourceExecutionReport('failed', f'http-{response.status}')
text = await response.text()
lines = text.splitlines()
for line in lines:
if hostname := normalize_scoped_hostname(line, self.word):
self.results.add(hostname)
if len(lines) >= requested and (self.limit is None or self.limit > self.PROVIDER_MAX_RESULTS):
return SourceExecutionReport('partial', 'provider-limit')
return None
except Exception as e:
error_msg = str(e).lower()
if '429' in error_msg or 'rate' in error_msg:
if attempt == self.max_retries - 1:
logger.info(f'THC rate limit failure after {self.max_retries} attempts')
return SourceExecutionReport('rate-limited', 'provider-rate-limit')
wait_time = self.base_delay * (attempt + 1)
logger.info(f'THC rate limit detected. Waiting {wait_time}s before retry...')
await asyncio.sleep(wait_time)
continue
logger.info(f'An exception has occurred in THC: {e}')
return SourceExecutionReport('failed', 'transport-error')
logger.info(f'An exception has occurred in THC: {e}')
return SourceExecutionReport('failed', 'transport-error')
except (aiohttp.ClientError, OSError, ValueError) as e:
logger.info(f'An exception has occurred in THC: {e}')
return SourceExecutionReport('failed', 'transport-error')
return SourceExecutionReport('failed', 'transport-error')
async def get_hostnames(self) -> set:
+2 -2
View File
@@ -10,7 +10,7 @@ from urllib.parse import urlsplit, urlunsplit
from theHarvester.discovery.constants import MissingKey
from theHarvester.discovery.provider_response import provider_http_error
from theHarvester.lib.core import AsyncFetcher, Core, FetcherResponse
from theHarvester.lib.hostnames import normalize_scoped_hostname
from theHarvester.lib.hostnames import normalize_hostname, normalize_scoped_hostname
from theHarvester.lib.source_execution import SourceExecutionReport, SourceReportStatus
from theHarvester.parsers import myparser
@@ -40,7 +40,7 @@ class SearchZoomEye:
if not isinstance(key, str) or not key.strip():
raise MissingKey('zoomeye')
self.word = word
self.target = word.strip().lower().removeprefix('www.').rstrip('.')
self.target = normalize_hostname(word)
self.limit = limit
self.key = key
self.baseurl = 'https://api.zoomeye.ai/v2/search'
+7 -1
View File
@@ -120,7 +120,13 @@ def parse_result_jsonl(payload: bytes | str) -> tuple[dict[str, object], list[di
if result_kind == 'prefix' and normalized_result_value != result_value:
raise ValueError('prefix result is not canonical')
except ValueError as error:
label = {'asn': 'ASN', 'prefix': 'prefix', 'shodan-host': 'Shodan host'}.get(str(result_kind), 'result')
label = {
'asn': 'ASN',
'hostname': 'hostname',
'ip': 'IP',
'prefix': 'prefix',
'shodan-host': 'Shodan host',
}.get(str(result_kind), 'result')
raise ValueError(f'JSONL findings must use a canonical {label} value') from error
record['value'] = normalized_result_value
if result_kind == 'prefix' and record.get('scope') != 'external-relationship':
+42 -54
View File
@@ -4,9 +4,11 @@ import asyncio
import contextlib
import json as json_loader
import logging
import os
import random
import re
import ssl
import tempfile
from dataclasses import dataclass
from pathlib import Path
from typing import TYPE_CHECKING, Any, ClassVar, Literal
@@ -218,8 +220,30 @@ class Core:
# Fallback to creating default in the user's home dir
default = (DATA_DIR / filename).read_text()
dest = CONFIG_DIRS[0].expanduser() / filename
dest.parent.mkdir(exist_ok=True)
dest.write_text(default)
dest.parent.mkdir(mode=0o700, exist_ok=True)
temporary_file = tempfile.NamedTemporaryFile(
mode='w',
encoding='utf-8',
dir=dest.parent,
prefix=f'.{filename}.',
delete=False,
)
temporary_path = Path(temporary_file.name)
try:
with temporary_file:
os.chmod(temporary_path, 0o600)
temporary_file.write(default)
temporary_file.flush()
os.fsync(temporary_file.fileno())
try:
os.link(temporary_path, dest)
except FileExistsError:
config = dest.read_text()
if not Core.quiet:
logger.info(f'Read {filename} from {dest}')
return config
finally:
temporary_path.unlink(missing_ok=True)
output_logger.info(f'Created default {filename} at {dest}')
return default
@@ -953,6 +977,8 @@ class AsyncFetcher:
session: aiohttp.ClientSession | None = None,
) -> list[Any]:
if session is not None:
if proxy:
raise ValueError('proxy selection is not supported with a caller-owned session')
return list(
await asyncio.gather(
*[
@@ -968,59 +994,21 @@ class AsyncFetcher:
)
)
# By default, timeout is 5 minutes; 60 seconds should suffice
headers = cls._default_headers(headers)
timeout = cls._request_timeout(60)
if len(params) == 0:
async with aiohttp.ClientSession(headers=headers, timeout=timeout) as session:
if proxy:
# Get random proxy for each URL (returns tuple of proxy_url and proxy_type)
proxy_data = [cls._get_random_proxy(cls().proxy_list) for _ in urls]
return list(
await asyncio.gather(
*[
AsyncFetcher.fetch(
session,
url,
json=json,
proxy=proxy_url,
include_metadata=include_metadata,
)
for url, (proxy_url, proxy_type) in zip(urls, proxy_data, strict=False)
]
async with cls.open_session(headers=headers, proxy=proxy, request_timeout=60) as owned_session:
return list(
await asyncio.gather(
*[
cls.fetch(
session=owned_session,
url=url,
params=params,
json=json,
include_metadata=include_metadata,
)
)
else:
return list(
await asyncio.gather(
*[AsyncFetcher.fetch(session, url, json=json, include_metadata=include_metadata) for url in urls]
)
)
else:
# Indicates the request has certain params
async with aiohttp.ClientSession(headers=headers, timeout=timeout) as session:
if proxy:
proxy_data = [cls._get_random_proxy(cls().proxy_list) for _ in urls]
return list(
await asyncio.gather(
*[
AsyncFetcher.fetch(
session,
url,
params,
json,
proxy=proxy_url,
include_metadata=include_metadata,
)
for url, (proxy_url, proxy_type) in zip(urls, proxy_data, strict=False)
]
)
)
else:
return list(
await asyncio.gather(
*[AsyncFetcher.fetch(session, url, params, json, include_metadata=include_metadata) for url in urls]
)
)
for url in urls
]
)
)
def show_default_error_message(engine_name: str, word: str, error) -> None:
+4 -3
View File
@@ -32,9 +32,10 @@ def normalize_scoped_hostname(value: object, target: str) -> str | None:
"""Return a canonical hostname when value is inside the target boundary."""
if not isinstance(value, str):
return None
hostname = value.strip().lower().rstrip('.')
normalized_target = target.strip().lower().rstrip('.')
if not normalized_target:
try:
hostname = normalize_hostname(value)
normalized_target = normalize_hostname(target)
except ValueError:
return None
if hostname == normalized_target or hostname.endswith(f'.{normalized_target}'):
return hostname
+15 -3
View File
@@ -36,16 +36,28 @@ def normalize_prefix(value: str) -> str:
raise ValueError('network prefix must be valid IPv4 or IPv6 CIDR') from error
def normalize_ip(value: str, *, label: str = 'IP result') -> str:
normalized = value.strip()
if '%' in normalized:
raise ValueError(f'{label} must not contain an IPv6 scope identifier')
try:
return str(ip_address(normalized))
except ValueError as error:
raise ValueError(f'{label} must be a valid IPv4 or IPv6 address') from error
def normalize_result_value(kind: ResultKind | str, value: str) -> str:
normalized = value.strip()
if kind == 'asn':
return normalize_asn(normalized)
if kind == 'hostname':
return normalize_hostname(normalized)
if kind == 'ip':
return normalize_ip(normalized)
if kind == 'prefix':
return normalize_prefix(normalized)
if kind == 'shodan-host':
if '%' in normalized:
raise ValueError('Shodan host must not contain an IPv6 scope identifier')
return str(ip_address(normalized))
return normalize_ip(normalized, label='Shodan host')
if kind == 'takeover':
return normalize_hostname(normalized)
return normalized
+2 -1
View File
@@ -246,5 +246,6 @@ def resolve_sources(selection: str | Iterable[str]) -> list[str]:
elif token in RESULT_CAPABILITIES:
selected.update(spec.name for spec in SOURCE_SPECS.values() if token in spec.capabilities)
else:
selected.add(token)
spec = _CASEFOLDED_SOURCE_SPECS.get(token.casefold())
selected.add(spec.name if spec is not None else token)
return sorted(selected)
+6 -6
View File
@@ -6,7 +6,6 @@ import logging
import time
from collections.abc import Awaitable, Callable, Iterable
from dataclasses import dataclass
from ipaddress import ip_address
from typing import TYPE_CHECKING, Any
from theHarvester.discovery import (
@@ -75,6 +74,7 @@ from theHarvester.lib.asn_attribution import AsnAttributionObservation, canonica
from theHarvester.lib.completed_result import ResultKind, ResultObservation, SourceExecution
from theHarvester.lib.enumeration import DEFAULT_SOURCE_WORKERS
from theHarvester.lib.hostnames import normalize_scoped_hostname
from theHarvester.lib.result_values import normalize_ip
from theHarvester.lib.shodan_evidence import ShodanHostObservation, canonical_shodan_hosts
from theHarvester.lib.source_catalog import ResultRoute, get_source_spec
from theHarvester.lib.source_execution import SourceExecutionReport
@@ -206,11 +206,11 @@ _BUILTWITH_GETTERS: tuple[tuple[str, ResultKind], ...] = (
def _normalize_values(request: SourceRequest, kind: ResultKind, values: Iterable[object]) -> set[ResultObservation]:
observations: set[ResultObservation] = set()
target = request.target.strip().lower().removeprefix('www.').rstrip('.')
canonical_target = normalize_scoped_hostname(request.target, request.target)
for item in values:
if kind == 'hostname':
value = normalize_scoped_hostname(item, target)
if value is None or value == target:
value = normalize_scoped_hostname(item, request.target)
if value is None or value == canonical_target:
continue
elif kind == 'email':
value = str(item).strip().lower()
@@ -218,7 +218,7 @@ def _normalize_values(request: SourceRequest, kind: ResultKind, values: Iterable
continue
elif kind == 'ip':
try:
value = str(ip_address(str(item).strip()))
value = normalize_ip(str(item))
except ValueError:
continue
elif kind in {'infostealer', 'person'}:
@@ -270,7 +270,7 @@ async def _collect_observations(
for host, address in await adapter.get_host_ip_pairs():
normalized_host = normalize_scoped_hostname(host, request.target)
try:
normalized_address = str(ip_address(address))
normalized_address = normalize_ip(str(address))
except ValueError:
continue
if (
+2 -4
View File
@@ -53,11 +53,10 @@ class Parser:
# Local part is required, charset is flexible.
# https://tools.ietf.org/html/rfc6531 (removed * and () as they provide FP mostly)
candidates = re.findall(r"[a-zA-Z0-9.\-_+#~!$&']+@[a-zA-Z0-9.-]+", self.results)
target = self.word.lower().removeprefix('www.')
emails: set[str] = set()
for candidate in candidates:
local_part, domain = candidate.lstrip('.').lower().split('@', maxsplit=1)
if normalized_domain := normalize_scoped_hostname(domain, target):
if normalized_domain := normalize_scoped_hostname(domain, self.word):
emails.add(f'{local_part}@{normalized_domain}')
return emails
@@ -75,10 +74,9 @@ class Parser:
async def hostnames(self):
await self.generic_clean()
target = self.word.lower().removeprefix('www.')
candidates = re.findall(r'[a-zA-Z0-9.-]+', self.results)
hostnames = {
normalized for candidate in candidates if (normalized := normalize_scoped_hostname(candidate.strip('.'), target))
normalized for candidate in candidates if (normalized := normalize_scoped_hostname(candidate.strip('.'), self.word))
}
return sorted(hostnames)