diff --git a/README.md b/README.md
index 318be873..f6e5f8ae 100644
--- a/README.md
+++ b/README.md
@@ -155,77 +155,70 @@ API clients send `THEHARVESTER_API_KEY` in the `X-API-Key` header. Provider API
Select sources by name or by a result route listed below. `-b all` runs the P0 sources. P1 and P2 sources require explicit selection.
-Result types in this table always appear in this order: `subdomains`, `emails`, `ips`, `asns`, `urls`, `people`, `breaches`. The first table contains sources that return only subdomains. The `API key` column refers to provider settings in `api-keys.yaml`; some providers require more than one value. `Optional` means the source can run without a key.
+Result types in this table always appear in this order: `subdomains`, `emails`, `ips`, `asns`, `urls`, `people`, `breaches`. A result followed by `only` means the source contributes no other result type. The `API key` column refers to provider settings in `api-keys.yaml`; some providers require more than one value. `Optional` means the source can run without a key.
The `shodan` source contributes subdomains. Shodan host enrichment through `-s` or `--shodan` is a separate action and is not a source result route.
-View all 58 sources by result type
-
-#### Subdomain-only sources (21)
-
-| Source | Activity | API key |
-| --- | :---: | :---: |
-| [`arquivo`](https://arquivo.pt/) | P0 | No |
-| [`certspotter`](https://sslmate.com/certspotter/) | P0 | No |
-| [`commoncrawl`](https://commoncrawl.org/) | P0 | No |
-| [`crt-name`](https://crt.name/) | P0 | No |
-| [`crtsh`](https://crt.sh/) | P0 | No |
-| [`dnsdb`](https://docs.domaintools.com/api/dnsdb/) | P0 | Required |
-| [`dymo`](https://docs.tpeoficial.com/docs/dymo-api/private/data-verifier) | P0 | Required |
-| [`fullhunt`](https://fullhunt.io/) | P0 | Required |
-| [`hunterhow`](https://hunter.how/) | P0 | Required |
-| [`leakix`](https://leakix.net/) | P0 | Required |
-| [`netlas`](https://netlas.io/) | P0 | Required |
-| [`projectdiscovery`](https://chaos.projectdiscovery.io/) | P0 | Required |
-| [`shodan`](https://www.shodan.io/) | P1 | Required |
-| [`shodanct`](https://ctl.shodan.io/) | P0 | No |
-| [`sourcegraph`](https://sourcegraph.com/search) | P0 | No |
-| [`subdomaincenter`](https://www.subdomain.center/) | P0 | No |
-| [`subdomainfinderc99`](https://subdomainfinder.c99.nl/) | P1 | No |
-| [`thc`](https://ip.thc.org/) | P0 | No |
-| [`virustotal`](https://www.virustotal.com/) | P0 | Required |
-| [`waybackarchive`](https://web.archive.org/) | P0 | No |
-| [`whoisxml`](https://subdomains.whoisxmlapi.com/) | P0 | Required |
-
-#### Sources that return other results (37)
+View all 58 discovery sources
| Source | Returns | Activity | API key |
| --- | --- | :---: | :---: |
| [`apis-guru`](https://apis.guru/) | subdomains, emails, urls | P0 | No |
+| [`arquivo`](https://arquivo.pt/) | subdomains only | P0 | No |
| [`baidu`](https://www.baidu.com/) | subdomains, emails | P0 | No |
| [`bevigil`](https://bevigil.com/osint-api) | subdomains, urls | P0 | Required |
| [`brave`](https://brave.com/search/api/) | subdomains, emails | P0 | Required |
| [`bufferoverun`](https://tls.bufferover.run/) | subdomains, ips | P0 | Required |
| [`builtwith`](https://builtwith.com/) | subdomains, urls | P0 | Required |
| [`censys`](https://search.censys.io/) | subdomains, emails | P0 | Required |
+| [`certspotter`](https://sslmate.com/certspotter/) | subdomains only | P0 | No |
+| [`commoncrawl`](https://commoncrawl.org/) | subdomains only | P0 | No |
| [`criminalip`](https://www.criminalip.io/) | subdomains, ips, asns | P2 | Required |
+| [`crt-name`](https://crt.name/) | subdomains only | P0 | No |
+| [`crtsh`](https://crt.sh/) | subdomains only | P0 | No |
| [`dehashed`](https://dehashed.com/) | emails, ips | P0 | Required |
+| [`dnsdb`](https://docs.domaintools.com/api/dnsdb/) | subdomains only | P0 | Required |
| [`dnsdumpster`](https://dnsdumpster.com/) | subdomains, ips | P0 | Required |
| [`duckduckgo`](https://duckduckgo.com/) | subdomains, emails | P0 | No |
+| [`dymo`](https://docs.tpeoficial.com/docs/dymo-api/private/data-verifier) | subdomains only | P0 | Required |
| [`fofa`](https://en.fofa.info/) | subdomains, ips | P0 | Required |
+| [`fullhunt`](https://fullhunt.io/) | subdomains only | P0 | Required |
| [`github-code`](https://github.com/) | subdomains, emails | P0 | Required |
| [`gitlab`](https://gitlab.com/) | subdomains, emails, urls | P0 | No |
| [`hackertarget`](https://hackertarget.com/) | subdomains, ips | P0 | Optional |
-| [`haveibeenpwned`](https://haveibeenpwned.com/) | breaches | P0 | No |
+| [`haveibeenpwned`](https://haveibeenpwned.com/) | breaches only | P0 | No |
| [`hibpverified`](https://haveibeenpwned.com/API/v3#BreachedDomain) | emails, breaches | P0 | Required |
| [`hudsonrock`](https://www.hudsonrock.com/) | subdomains, emails, ips | P0 | No |
| [`hunter`](https://hunter.io/) | subdomains, emails | P0 | Required |
+| [`hunterhow`](https://hunter.how/) | subdomains only | P0 | Required |
| [`intelx`](https://intelx.io/) | subdomains, emails, urls | P0 | Required |
+| [`leakix`](https://leakix.net/) | subdomains only | P0 | Required |
| [`leaklookup`](https://leak-lookup.com/) | emails, breaches | P0 | Required |
| [`mojeek`](https://www.mojeek.com/services/search/web-search-api/) | subdomains, emails | P0 | Optional |
+| [`netlas`](https://netlas.io/) | subdomains only | P0 | Required |
| [`onyphe`](https://www.onyphe.io/) | subdomains, ips, asns | P0 | Required |
| [`otx`](https://otx.alienvault.com/) | subdomains, ips | P0 | No |
| [`pentesttools`](https://pentest-tools.com/) | subdomains, ips | P1 | Required |
+| [`projectdiscovery`](https://chaos.projectdiscovery.io/) | subdomains only | P0 | Required |
| [`rapiddns`](https://rapiddns.io/) | subdomains, ips | P0 | No |
-| [`robtex`](https://www.robtex.com/) | ips | P0 | No |
+| [`robtex`](https://www.robtex.com/) | ips only | P0 | No |
| [`rocketreach`](https://rocketreach.co/) | emails, urls | P0 | Required |
| [`securityscorecard`](https://securityscorecard.com/) | subdomains, ips | P0 | Required |
| [`securityTrails`](https://securitytrails.com/) | subdomains, ips | P0 | Required |
| [`sherlockeye`](https://sherlockeye.io/) | subdomains, emails, ips | P0 | Required |
+| [`shodan`](https://www.shodan.io/) | subdomains only | P1 | Required |
+| [`shodanct`](https://ctl.shodan.io/) | subdomains only | P0 | No |
| [`shodanInternetDB`](https://internetdb.shodan.io/) | subdomains, ips | P1 | No |
+| [`sourcegraph`](https://sourcegraph.com/search) | subdomains only | P0 | No |
+| [`subdomaincenter`](https://www.subdomain.center/) | subdomains only | P0 | No |
+| [`subdomainfinderc99`](https://subdomainfinder.c99.nl/) | subdomains only | P1 | No |
+| [`thc`](https://ip.thc.org/) | subdomains only | P0 | No |
| [`tomba`](https://tomba.io/) | subdomains, emails | P0 | Required |
| [`urlscan`](https://urlscan.io/) | subdomains, ips, asns, urls | P0 | No |
+| [`virustotal`](https://www.virustotal.com/) | subdomains only | P0 | Required |
+| [`waybackarchive`](https://web.archive.org/) | subdomains only | P0 | No |
+| [`whoisxml`](https://subdomains.whoisxmlapi.com/) | subdomains only | P0 | Required |
| [`windvane`](https://windvane.lichoin.com/) | subdomains, emails, ips | P0 | Optional |
| [`yahoo`](https://www.yahoo.com/) | subdomains, emails | P0 | No |
| [`zoomeye`](https://www.zoomeye.ai/) | subdomains, emails, ips, asns, urls | P0 | Required |
@@ -255,9 +248,16 @@ Treat collected OSINT as potentially sensitive. Keep report files, screenshots,
JSONL is the primary format for automation and one-run interchange. The first line describes the run. Each remaining line is one sorted, deduplicated finding with its source and action provenance.
+This example contains six findings from two sources. The `counts` object summarizes the result lines that follow it.
+
```jsonl
-{"action_executions":[],"artifacts":[],"completed_at":"2026-08-07T12:01:00Z","counts":{"hostname":1},"evidence_status":"complete","result_count":1,"run_id":"123e4567-e89b-12d3-a456-426614174000","source_executions":[],"started_at":"2026-08-07T12:00:00Z","target":"example.com","type":"summary"}
-{"sources":[],"type":"hostname","value":"api.example.com"}
+{"action_executions":[],"artifacts":[],"completed_at":"2026-08-17T12:01:00Z","counts":{"asn":1,"breach":1,"email":1,"hostname":1,"ip":1,"url":1},"evidence_status":"complete","result_count":6,"run_id":"123e4567-e89b-12d3-a456-426614174000","source_executions":[{"duration_ms":127.4,"error_type":null,"result_count":1,"source":"haveibeenpwned","status":"completed","stop_reason":null},{"duration_ms":482.3,"error_type":null,"result_count":5,"source":"zoomeye","status":"completed","stop_reason":null}],"started_at":"2026-08-17T12:00:00Z","target":"example.com","type":"summary"}
+{"sources":["zoomeye"],"type":"asn","value":"AS64500"}
+{"sources":["haveibeenpwned"],"type":"breach","value":"Example breach"}
+{"sources":["zoomeye"],"type":"email","value":"security@example.com"}
+{"sources":["zoomeye"],"type":"hostname","value":"api.example.com"}
+{"sources":["zoomeye"],"type":"ip","value":"192.0.2.10"}
+{"sources":["zoomeye"],"type":"url","value":"https://api.example.com/login"}
```
Extract common result types with `jq`:
diff --git a/tests/test_readme.py b/tests/test_readme.py
index 7bae2860..cadc1cf3 100644
--- a/tests/test_readme.py
+++ b/tests/test_readme.py
@@ -6,6 +6,7 @@ from pathlib import Path
import yaml
+from theHarvester.lib.completed_result import parse_result_jsonl
from theHarvester.lib.source_catalog import ACTION_ACTIVITIES, RESULT_CAPABILITIES, SOURCE_SPECS
OPTIONAL_API_KEY_SOURCES = {'hackertarget', 'mojeek', 'windvane'}
@@ -99,7 +100,7 @@ def _declared_source_contracts() -> dict[str, set[str]]:
def _source_matrix(readme: str) -> str:
- return readme.split('View all 58 sources by result type
', 1)[1].split(' ', 1)[0]
+ return readme.split('View all 58 discovery sources', 1)[1].split('', 1)[0]
def _documented_source_rows(readme: str) -> dict[str, list[str]]:
@@ -109,8 +110,7 @@ def _documented_source_rows(readme: str) -> dict[str, list[str]]:
cells = [cell.strip() for cell in line.strip('|').split('|')]
source = re.fullmatch(r'\[`([^`]+)`\]\((https://[^)]+)\)', cells[0])
assert source is not None
- values = cells[1:]
- rows[source.group(1)] = ['subdomains', *values] if len(values) == 2 else values
+ rows[source.group(1)] = cells[1:]
return rows
@@ -123,7 +123,10 @@ def _documented_source_links(readme: str) -> list[tuple[str, str]]:
def _documented_source_contracts(readme: str) -> dict[str, set[str]]:
- return {source: {route.strip() for route in cells[0].split(',')} for source, cells in _documented_source_rows(readme).items()}
+ return {
+ source: {route.strip().removesuffix(' only') for route in cells[0].split(',')}
+ for source, cells in _documented_source_rows(readme).items()
+ }
def _documented_source_activities(readme: str) -> dict[str, str]:
@@ -144,8 +147,7 @@ def test_readme_matches_declared_source_contracts() -> None:
documented = _documented_source_contracts(readme)
declared = _declared_source_contracts()
- assert '| Source | Activity | API key |' in readme
- assert '| Source | Returns | Activity | API key |' in readme
+ assert _source_matrix(readme).count('| Source | Returns | Activity | API key |') == 1
assert 'Credentials |' not in _source_matrix(readme)
assert len(declared) == 58
assert len(documented) == 58
@@ -157,29 +159,21 @@ def test_readme_matches_declared_source_contracts() -> None:
assert {'securitytrails', 'shodaninternetdb'}.isdisjoint(documented)
-def test_readme_source_matrix_is_grouped_and_ordered() -> None:
+def test_readme_source_matrix_is_one_table_and_ordered() -> None:
readme = Path('README.md').read_text()
matrix = _source_matrix(readme)
- subdomain_section, other_section = matrix.split('#### Sources that return other results (37)', 1)
+ rows = _documented_source_rows(readme)
+ names = list(rows)
- def names(section: str) -> list[str]:
- return [match.group(1) for line in section.splitlines() if (match := re.match(r'^\| \[`([^`]+)`\]\(https://', line))]
-
- subdomain_only = names(subdomain_section)
- other_results = names(other_section)
- declared_subdomain_only = {source for source, spec in SOURCE_SPECS.items() if spec.capabilities == frozenset({'subdomains'})}
-
- assert subdomain_only == sorted(subdomain_only, key=str.casefold)
- assert other_results == sorted(other_results, key=str.casefold)
- assert len(subdomain_only) == 21
- assert len(other_results) == 37
- assert set(subdomain_only) == declared_subdomain_only
- assert set(other_results) == set(SOURCE_SPECS) - declared_subdomain_only
+ assert '#### Subdomain-only sources' not in matrix
+ assert '#### Sources that return other results' not in matrix
+ assert names == sorted(names, key=str.casefold)
route_rank = {route: index for index, route in enumerate(SOURCE_ROUTE_ORDER)}
- for routes, *_ in _documented_source_rows(readme).values():
- documented_order = [route.strip() for route in routes.split(',')]
+ for source, (routes, *_) in rows.items():
+ documented_order = [route.strip().removesuffix(' only') for route in routes.split(',')]
assert documented_order == sorted(documented_order, key=route_rank.__getitem__)
+ assert routes.endswith(' only') == (len(SOURCE_SPECS[source].capabilities) == 1)
def test_readme_api_key_markers_match_configuration() -> None:
@@ -364,8 +358,16 @@ def test_operator_docs_cover_portable_database_export() -> None:
def test_readme_explains_jsonl_record_and_structured_evidence_parsing() -> None:
readme = Path('README.md').read_text()
+ example = re.search(r'```jsonl\n(.*?)\n```', readme, flags=re.DOTALL)
+ assert example is not None
+ summary, findings = parse_result_jsonl(example.group(1))
- assert '{"sources":[],"type":"hostname","value":"api.example.com"}' in readme
+ finding_types = ('asn', 'breach', 'email', 'hostname', 'ip', 'url')
+ assert summary['counts'] == {result_type: 1 for result_type in finding_types}
+ assert summary['result_count'] == len(findings) == 6
+ assert tuple(finding['type'] for finding in findings) == finding_types
+ assert all(finding['sources'] for finding in findings)
+ assert {execution['source'] for execution in summary['source_executions']} == {'haveibeenpwned', 'zoomeye'}
for result_kind in ('hostname', 'ip', 'asn', 'email', 'url', 'person', 'breach'):
assert f'select(.type == "{result_kind}")' in readme
assert 'select(.type == "person") | .value | fromjson' in readme