diff --git a/README.md b/README.md index 318be873..f6e5f8ae 100644 --- a/README.md +++ b/README.md @@ -155,77 +155,70 @@ API clients send `THEHARVESTER_API_KEY` in the `X-API-Key` header. Provider API Select sources by name or by a result route listed below. `-b all` runs the P0 sources. P1 and P2 sources require explicit selection. -Result types in this table always appear in this order: `subdomains`, `emails`, `ips`, `asns`, `urls`, `people`, `breaches`. The first table contains sources that return only subdomains. The `API key` column refers to provider settings in `api-keys.yaml`; some providers require more than one value. `Optional` means the source can run without a key. +Result types in this table always appear in this order: `subdomains`, `emails`, `ips`, `asns`, `urls`, `people`, `breaches`. A result followed by `only` means the source contributes no other result type. The `API key` column refers to provider settings in `api-keys.yaml`; some providers require more than one value. `Optional` means the source can run without a key. The `shodan` source contributes subdomains. Shodan host enrichment through `-s` or `--shodan` is a separate action and is not a source result route.
-View all 58 sources by result type - -#### Subdomain-only sources (21) - -| Source | Activity | API key | -| --- | :---: | :---: | -| [`arquivo`](https://arquivo.pt/) | P0 | No | -| [`certspotter`](https://sslmate.com/certspotter/) | P0 | No | -| [`commoncrawl`](https://commoncrawl.org/) | P0 | No | -| [`crt-name`](https://crt.name/) | P0 | No | -| [`crtsh`](https://crt.sh/) | P0 | No | -| [`dnsdb`](https://docs.domaintools.com/api/dnsdb/) | P0 | Required | -| [`dymo`](https://docs.tpeoficial.com/docs/dymo-api/private/data-verifier) | P0 | Required | -| [`fullhunt`](https://fullhunt.io/) | P0 | Required | -| [`hunterhow`](https://hunter.how/) | P0 | Required | -| [`leakix`](https://leakix.net/) | P0 | Required | -| [`netlas`](https://netlas.io/) | P0 | Required | -| [`projectdiscovery`](https://chaos.projectdiscovery.io/) | P0 | Required | -| [`shodan`](https://www.shodan.io/) | P1 | Required | -| [`shodanct`](https://ctl.shodan.io/) | P0 | No | -| [`sourcegraph`](https://sourcegraph.com/search) | P0 | No | -| [`subdomaincenter`](https://www.subdomain.center/) | P0 | No | -| [`subdomainfinderc99`](https://subdomainfinder.c99.nl/) | P1 | No | -| [`thc`](https://ip.thc.org/) | P0 | No | -| [`virustotal`](https://www.virustotal.com/) | P0 | Required | -| [`waybackarchive`](https://web.archive.org/) | P0 | No | -| [`whoisxml`](https://subdomains.whoisxmlapi.com/) | P0 | Required | - -#### Sources that return other results (37) +View all 58 discovery sources | Source | Returns | Activity | API key | | --- | --- | :---: | :---: | | [`apis-guru`](https://apis.guru/) | subdomains, emails, urls | P0 | No | +| [`arquivo`](https://arquivo.pt/) | subdomains only | P0 | No | | [`baidu`](https://www.baidu.com/) | subdomains, emails | P0 | No | | [`bevigil`](https://bevigil.com/osint-api) | subdomains, urls | P0 | Required | | [`brave`](https://brave.com/search/api/) | subdomains, emails | P0 | Required | | [`bufferoverun`](https://tls.bufferover.run/) | subdomains, ips | P0 | Required | | [`builtwith`](https://builtwith.com/) | subdomains, urls | P0 | Required | | [`censys`](https://search.censys.io/) | subdomains, emails | P0 | Required | +| [`certspotter`](https://sslmate.com/certspotter/) | subdomains only | P0 | No | +| [`commoncrawl`](https://commoncrawl.org/) | subdomains only | P0 | No | | [`criminalip`](https://www.criminalip.io/) | subdomains, ips, asns | P2 | Required | +| [`crt-name`](https://crt.name/) | subdomains only | P0 | No | +| [`crtsh`](https://crt.sh/) | subdomains only | P0 | No | | [`dehashed`](https://dehashed.com/) | emails, ips | P0 | Required | +| [`dnsdb`](https://docs.domaintools.com/api/dnsdb/) | subdomains only | P0 | Required | | [`dnsdumpster`](https://dnsdumpster.com/) | subdomains, ips | P0 | Required | | [`duckduckgo`](https://duckduckgo.com/) | subdomains, emails | P0 | No | +| [`dymo`](https://docs.tpeoficial.com/docs/dymo-api/private/data-verifier) | subdomains only | P0 | Required | | [`fofa`](https://en.fofa.info/) | subdomains, ips | P0 | Required | +| [`fullhunt`](https://fullhunt.io/) | subdomains only | P0 | Required | | [`github-code`](https://github.com/) | subdomains, emails | P0 | Required | | [`gitlab`](https://gitlab.com/) | subdomains, emails, urls | P0 | No | | [`hackertarget`](https://hackertarget.com/) | subdomains, ips | P0 | Optional | -| [`haveibeenpwned`](https://haveibeenpwned.com/) | breaches | P0 | No | +| [`haveibeenpwned`](https://haveibeenpwned.com/) | breaches only | P0 | No | | [`hibpverified`](https://haveibeenpwned.com/API/v3#BreachedDomain) | emails, breaches | P0 | Required | | [`hudsonrock`](https://www.hudsonrock.com/) | subdomains, emails, ips | P0 | No | | [`hunter`](https://hunter.io/) | subdomains, emails | P0 | Required | +| [`hunterhow`](https://hunter.how/) | subdomains only | P0 | Required | | [`intelx`](https://intelx.io/) | subdomains, emails, urls | P0 | Required | +| [`leakix`](https://leakix.net/) | subdomains only | P0 | Required | | [`leaklookup`](https://leak-lookup.com/) | emails, breaches | P0 | Required | | [`mojeek`](https://www.mojeek.com/services/search/web-search-api/) | subdomains, emails | P0 | Optional | +| [`netlas`](https://netlas.io/) | subdomains only | P0 | Required | | [`onyphe`](https://www.onyphe.io/) | subdomains, ips, asns | P0 | Required | | [`otx`](https://otx.alienvault.com/) | subdomains, ips | P0 | No | | [`pentesttools`](https://pentest-tools.com/) | subdomains, ips | P1 | Required | +| [`projectdiscovery`](https://chaos.projectdiscovery.io/) | subdomains only | P0 | Required | | [`rapiddns`](https://rapiddns.io/) | subdomains, ips | P0 | No | -| [`robtex`](https://www.robtex.com/) | ips | P0 | No | +| [`robtex`](https://www.robtex.com/) | ips only | P0 | No | | [`rocketreach`](https://rocketreach.co/) | emails, urls | P0 | Required | | [`securityscorecard`](https://securityscorecard.com/) | subdomains, ips | P0 | Required | | [`securityTrails`](https://securitytrails.com/) | subdomains, ips | P0 | Required | | [`sherlockeye`](https://sherlockeye.io/) | subdomains, emails, ips | P0 | Required | +| [`shodan`](https://www.shodan.io/) | subdomains only | P1 | Required | +| [`shodanct`](https://ctl.shodan.io/) | subdomains only | P0 | No | | [`shodanInternetDB`](https://internetdb.shodan.io/) | subdomains, ips | P1 | No | +| [`sourcegraph`](https://sourcegraph.com/search) | subdomains only | P0 | No | +| [`subdomaincenter`](https://www.subdomain.center/) | subdomains only | P0 | No | +| [`subdomainfinderc99`](https://subdomainfinder.c99.nl/) | subdomains only | P1 | No | +| [`thc`](https://ip.thc.org/) | subdomains only | P0 | No | | [`tomba`](https://tomba.io/) | subdomains, emails | P0 | Required | | [`urlscan`](https://urlscan.io/) | subdomains, ips, asns, urls | P0 | No | +| [`virustotal`](https://www.virustotal.com/) | subdomains only | P0 | Required | +| [`waybackarchive`](https://web.archive.org/) | subdomains only | P0 | No | +| [`whoisxml`](https://subdomains.whoisxmlapi.com/) | subdomains only | P0 | Required | | [`windvane`](https://windvane.lichoin.com/) | subdomains, emails, ips | P0 | Optional | | [`yahoo`](https://www.yahoo.com/) | subdomains, emails | P0 | No | | [`zoomeye`](https://www.zoomeye.ai/) | subdomains, emails, ips, asns, urls | P0 | Required | @@ -255,9 +248,16 @@ Treat collected OSINT as potentially sensitive. Keep report files, screenshots, JSONL is the primary format for automation and one-run interchange. The first line describes the run. Each remaining line is one sorted, deduplicated finding with its source and action provenance. +This example contains six findings from two sources. The `counts` object summarizes the result lines that follow it. + ```jsonl -{"action_executions":[],"artifacts":[],"completed_at":"2026-08-07T12:01:00Z","counts":{"hostname":1},"evidence_status":"complete","result_count":1,"run_id":"123e4567-e89b-12d3-a456-426614174000","source_executions":[],"started_at":"2026-08-07T12:00:00Z","target":"example.com","type":"summary"} -{"sources":[],"type":"hostname","value":"api.example.com"} +{"action_executions":[],"artifacts":[],"completed_at":"2026-08-17T12:01:00Z","counts":{"asn":1,"breach":1,"email":1,"hostname":1,"ip":1,"url":1},"evidence_status":"complete","result_count":6,"run_id":"123e4567-e89b-12d3-a456-426614174000","source_executions":[{"duration_ms":127.4,"error_type":null,"result_count":1,"source":"haveibeenpwned","status":"completed","stop_reason":null},{"duration_ms":482.3,"error_type":null,"result_count":5,"source":"zoomeye","status":"completed","stop_reason":null}],"started_at":"2026-08-17T12:00:00Z","target":"example.com","type":"summary"} +{"sources":["zoomeye"],"type":"asn","value":"AS64500"} +{"sources":["haveibeenpwned"],"type":"breach","value":"Example breach"} +{"sources":["zoomeye"],"type":"email","value":"security@example.com"} +{"sources":["zoomeye"],"type":"hostname","value":"api.example.com"} +{"sources":["zoomeye"],"type":"ip","value":"192.0.2.10"} +{"sources":["zoomeye"],"type":"url","value":"https://api.example.com/login"} ``` Extract common result types with `jq`: diff --git a/tests/test_readme.py b/tests/test_readme.py index 7bae2860..cadc1cf3 100644 --- a/tests/test_readme.py +++ b/tests/test_readme.py @@ -6,6 +6,7 @@ from pathlib import Path import yaml +from theHarvester.lib.completed_result import parse_result_jsonl from theHarvester.lib.source_catalog import ACTION_ACTIVITIES, RESULT_CAPABILITIES, SOURCE_SPECS OPTIONAL_API_KEY_SOURCES = {'hackertarget', 'mojeek', 'windvane'} @@ -99,7 +100,7 @@ def _declared_source_contracts() -> dict[str, set[str]]: def _source_matrix(readme: str) -> str: - return readme.split('View all 58 sources by result type', 1)[1].split('
', 1)[0] + return readme.split('View all 58 discovery sources', 1)[1].split('', 1)[0] def _documented_source_rows(readme: str) -> dict[str, list[str]]: @@ -109,8 +110,7 @@ def _documented_source_rows(readme: str) -> dict[str, list[str]]: cells = [cell.strip() for cell in line.strip('|').split('|')] source = re.fullmatch(r'\[`([^`]+)`\]\((https://[^)]+)\)', cells[0]) assert source is not None - values = cells[1:] - rows[source.group(1)] = ['subdomains', *values] if len(values) == 2 else values + rows[source.group(1)] = cells[1:] return rows @@ -123,7 +123,10 @@ def _documented_source_links(readme: str) -> list[tuple[str, str]]: def _documented_source_contracts(readme: str) -> dict[str, set[str]]: - return {source: {route.strip() for route in cells[0].split(',')} for source, cells in _documented_source_rows(readme).items()} + return { + source: {route.strip().removesuffix(' only') for route in cells[0].split(',')} + for source, cells in _documented_source_rows(readme).items() + } def _documented_source_activities(readme: str) -> dict[str, str]: @@ -144,8 +147,7 @@ def test_readme_matches_declared_source_contracts() -> None: documented = _documented_source_contracts(readme) declared = _declared_source_contracts() - assert '| Source | Activity | API key |' in readme - assert '| Source | Returns | Activity | API key |' in readme + assert _source_matrix(readme).count('| Source | Returns | Activity | API key |') == 1 assert 'Credentials |' not in _source_matrix(readme) assert len(declared) == 58 assert len(documented) == 58 @@ -157,29 +159,21 @@ def test_readme_matches_declared_source_contracts() -> None: assert {'securitytrails', 'shodaninternetdb'}.isdisjoint(documented) -def test_readme_source_matrix_is_grouped_and_ordered() -> None: +def test_readme_source_matrix_is_one_table_and_ordered() -> None: readme = Path('README.md').read_text() matrix = _source_matrix(readme) - subdomain_section, other_section = matrix.split('#### Sources that return other results (37)', 1) + rows = _documented_source_rows(readme) + names = list(rows) - def names(section: str) -> list[str]: - return [match.group(1) for line in section.splitlines() if (match := re.match(r'^\| \[`([^`]+)`\]\(https://', line))] - - subdomain_only = names(subdomain_section) - other_results = names(other_section) - declared_subdomain_only = {source for source, spec in SOURCE_SPECS.items() if spec.capabilities == frozenset({'subdomains'})} - - assert subdomain_only == sorted(subdomain_only, key=str.casefold) - assert other_results == sorted(other_results, key=str.casefold) - assert len(subdomain_only) == 21 - assert len(other_results) == 37 - assert set(subdomain_only) == declared_subdomain_only - assert set(other_results) == set(SOURCE_SPECS) - declared_subdomain_only + assert '#### Subdomain-only sources' not in matrix + assert '#### Sources that return other results' not in matrix + assert names == sorted(names, key=str.casefold) route_rank = {route: index for index, route in enumerate(SOURCE_ROUTE_ORDER)} - for routes, *_ in _documented_source_rows(readme).values(): - documented_order = [route.strip() for route in routes.split(',')] + for source, (routes, *_) in rows.items(): + documented_order = [route.strip().removesuffix(' only') for route in routes.split(',')] assert documented_order == sorted(documented_order, key=route_rank.__getitem__) + assert routes.endswith(' only') == (len(SOURCE_SPECS[source].capabilities) == 1) def test_readme_api_key_markers_match_configuration() -> None: @@ -364,8 +358,16 @@ def test_operator_docs_cover_portable_database_export() -> None: def test_readme_explains_jsonl_record_and_structured_evidence_parsing() -> None: readme = Path('README.md').read_text() + example = re.search(r'```jsonl\n(.*?)\n```', readme, flags=re.DOTALL) + assert example is not None + summary, findings = parse_result_jsonl(example.group(1)) - assert '{"sources":[],"type":"hostname","value":"api.example.com"}' in readme + finding_types = ('asn', 'breach', 'email', 'hostname', 'ip', 'url') + assert summary['counts'] == {result_type: 1 for result_type in finding_types} + assert summary['result_count'] == len(findings) == 6 + assert tuple(finding['type'] for finding in findings) == finding_types + assert all(finding['sources'] for finding in findings) + assert {execution['source'] for execution in summary['source_executions']} == {'haveibeenpwned', 'zoomeye'} for result_kind in ('hostname', 'ip', 'asn', 'email', 'url', 'person', 'breach'): assert f'select(.type == "{result_kind}")' in readme assert 'select(.type == "person") | .value | fromjson' in readme