diff --git a/restfulHarvest.py b/restfulHarvest.py
index 7cc31812..d7b02cf7 100755
--- a/restfulHarvest.py
+++ b/restfulHarvest.py
@@ -8,7 +8,7 @@ parser.add_argument('-p', '--port', default=5000, help='Port to bind the web ser
parser.add_argument('-l', '--log-level', default='info', help='Set logging level, default is info but [critical|error|warning|info|debug|trace] can be set')
parser.add_argument('-r', '--reload', default=False, help='Enable automatic reload used during development of the api', action='store_true')
-args = parser.parse_args()
+args: argparse.Namespace = parser.parse_args()
if __name__ == '__main__':
uvicorn.run('theHarvester.lib.api.api:app', host=args.host, port=args.port, log_level=args.log_level, reload=args.reload)
diff --git a/setup.py b/setup.py
index d09dea9c..dd01fe6c 100755
--- a/setup.py
+++ b/setup.py
@@ -2,7 +2,7 @@ from setuptools import setup, find_packages
from theHarvester.lib.core import Core
with open('README.md', 'r') as fh:
- long_description = fh.read()
+ long_description: str = fh.read()
setup(
name='theHarvester',
@@ -20,9 +20,8 @@ setup(
classifiers=[
"Programming Language :: Python :: 3",
- "Programming Language :: Python :: 3.8",
- "Programming Language :: Python :: 3.9",
"Programming Language :: Python :: 3.10",
+ "Programming Language :: Python :: 3.11",
"License :: OSI Approved :: GNU General Public License v2 (GPLv2)",
"Operating System :: OS Independent",
],
diff --git a/tests/discovery/test_anubis.py b/tests/discovery/test_anubis.py
index 684ae85c..1b1d492b 100644
--- a/tests/discovery/test_anubis.py
+++ b/tests/discovery/test_anubis.py
@@ -5,9 +5,11 @@ from theHarvester.lib.core import *
from theHarvester.discovery import anubis
import os
import pytest
+from _pytest.mark.structures import MarkDecorator
+from typing import Optional
-pytestmark = pytest.mark.asyncio
-github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
+pytestmark: MarkDecorator = pytest.mark.asyncio
+github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
class TestAnubis:
@@ -15,7 +17,7 @@ class TestAnubis:
def domain() -> str:
return 'apple.com'
- async def test_api(self):
+ async def test_api(self) -> None:
base_url = f'https://jldc.me/anubis/subdomains/{TestAnubis.domain()}'
headers = {'User-Agent': Core.get_user_agent()}
request = requests.get(base_url, headers=headers)
@@ -26,6 +28,6 @@ class TestAnubis:
await search.do_search()
return await search.get_hostnames()
- async def test_process(self):
+ async def test_process(self) -> None:
await self.test_do_search()
assert len(await self.test_do_search()) > 0
diff --git a/tests/discovery/test_certspotter.py b/tests/discovery/test_certspotter.py
index aa73e39b..19152693 100644
--- a/tests/discovery/test_certspotter.py
+++ b/tests/discovery/test_certspotter.py
@@ -5,9 +5,11 @@ from theHarvester.discovery import certspottersearch
import os
import requests
import pytest
+from _pytest.mark.structures import MarkDecorator
+from typing import Optional
-pytestmark = pytest.mark.asyncio
-github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
+pytestmark: MarkDecorator = pytest.mark.asyncio
+github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
class TestCertspotter(object):
@@ -15,18 +17,18 @@ class TestCertspotter(object):
def domain() -> str:
return 'metasploit.com'
- async def test_api(self):
+ async def test_api(self) -> None:
base_url = f'https://api.certspotter.com/v1/issuances?domain={TestCertspotter.domain()}&expand=dns_names'
headers = {'User-Agent': Core.get_user_agent()}
request = requests.get(base_url, headers=headers)
assert request.status_code == 200
- async def test_search(self):
+ async def test_search(self) -> None:
search = certspottersearch.SearchCertspoter(TestCertspotter.domain())
await search.process()
assert isinstance(await search.get_hostnames(), set)
- async def test_search_no_results(self):
+ async def test_search_no_results(self) -> None:
search = certspottersearch.SearchCertspoter('radiant.eu')
await search.process()
assert len(await search.get_hostnames()) == 0
diff --git a/tests/discovery/test_githubcode.py b/tests/discovery/test_githubcode.py
index 0e7d52d6..0eb9154d 100644
--- a/tests/discovery/test_githubcode.py
+++ b/tests/discovery/test_githubcode.py
@@ -4,8 +4,9 @@ from theHarvester.lib.core import Core
from unittest.mock import MagicMock
from requests import Response
import pytest
+from _pytest.mark.structures import MarkDecorator
-pytestmark = pytest.mark.asyncio
+pytestmark: MarkDecorator = pytest.mark.asyncio
class TestSearchGithubCode:
@@ -65,31 +66,31 @@ class TestSearchGithubCode:
response.json = MagicMock(return_value=json)
response.status_code = 200
- async def test_missing_key(self):
+ async def test_missing_key(self) -> None:
with pytest.raises(MissingKey):
Core.github_key = MagicMock(return_value=None)
githubcode.SearchGithubCode(word="test", limit=500)
- async def test_fragments_from_response(self):
+ async def test_fragments_from_response(self) -> None:
Core.github_key = MagicMock(return_value="lol")
test_class_instance = githubcode.SearchGithubCode(word="test", limit=500)
test_result = await test_class_instance.fragments_from_response(self.OkResponse.response.json())
print('test_result: ', test_result)
assert test_result == ["test1", "test2"]
- async def test_invalid_fragments_from_response(self):
+ async def test_invalid_fragments_from_response(self) -> None:
Core.github_key = MagicMock(return_value="lol")
test_class_instance = githubcode.SearchGithubCode(word="test", limit=500)
test_result = await test_class_instance.fragments_from_response(self.MalformedResponse.response.json())
assert test_result == []
- async def test_next_page(self):
+ async def test_next_page(self) -> None:
Core.github_key = MagicMock(return_value="lol")
test_class_instance = githubcode.SearchGithubCode(word="test", limit=500)
test_result = githubcode.SuccessResult(list(), next_page=2, last_page=4)
assert (2 == await test_class_instance.next_page_or_end(test_result))
- async def test_last_page(self):
+ async def test_last_page(self) -> None:
Core.github_key = MagicMock(return_value="lol")
test_class_instance = githubcode.SearchGithubCode(word="test", limit=500)
test_result = githubcode.SuccessResult(list(), None, None)
diff --git a/tests/discovery/test_omnisint.py b/tests/discovery/test_omnisint.py
index bc492ee0..3277be2c 100644
--- a/tests/discovery/test_omnisint.py
+++ b/tests/discovery/test_omnisint.py
@@ -5,9 +5,11 @@ from theHarvester.discovery import omnisint
import os
import requests
import pytest
+from _pytest.mark.structures import MarkDecorator
+from typing import Optional
-pytestmark = pytest.mark.asyncio
-github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
+pytestmark: MarkDecorator = pytest.mark.asyncio
+github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
class TestOmnisint(object):
@@ -16,13 +18,13 @@ class TestOmnisint(object):
return 'uber.com'
@pytest.mark.skipif(github_ci == 'true', reason='Skipping on Github CI due to unstable status code from site')
- async def test_api(self):
+ async def test_api(self) -> None:
base_url = f'https://sonar.omnisint.io/all/{TestOmnisint.domain()}'
headers = {'User-Agent': Core.get_user_agent()}
request = requests.get(base_url, headers=headers)
assert request.status_code == 200
- async def test_search(self):
+ async def test_search(self) -> None:
search = omnisint.SearchOmnisint(TestOmnisint.domain())
await search.process()
assert isinstance(await search.get_hostnames(), list)
diff --git a/tests/discovery/test_otx.py b/tests/discovery/test_otx.py
index acc41c4c..851afb9d 100644
--- a/tests/discovery/test_otx.py
+++ b/tests/discovery/test_otx.py
@@ -5,9 +5,11 @@ from theHarvester.discovery import otxsearch
import os
import requests
import pytest
+from _pytest.mark.structures import MarkDecorator
+from typing import Optional
-pytestmark = pytest.mark.asyncio
-github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
+pytestmark: MarkDecorator = pytest.mark.asyncio
+github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
class TestOtx(object):
@@ -15,13 +17,13 @@ class TestOtx(object):
def domain() -> str:
return 'metasploit.com'
- async def test_api(self):
+ async def test_api(self) -> None:
base_url = f'https://otx.alienvault.com/api/v1/indicators/domain/{TestOtx.domain()}/passive_dns'
headers = {'User-Agent': Core.get_user_agent()}
request = requests.get(base_url, headers=headers)
assert request.status_code == 200
- async def test_search(self):
+ async def test_search(self) -> None:
search = otxsearch.SearchOtx(TestOtx.domain())
await search.process()
assert isinstance(await search.get_hostnames(), set)
diff --git a/tests/discovery/test_qwantsearch.py b/tests/discovery/test_qwantsearch.py
index 2fdcad4d..7ce21f75 100644
--- a/tests/discovery/test_qwantsearch.py
+++ b/tests/discovery/test_qwantsearch.py
@@ -3,9 +3,11 @@
from theHarvester.discovery import qwantsearch
import os
import pytest
+from _pytest.mark.structures import MarkDecorator
+from typing import Optional
-pytestmark = pytest.mark.asyncio
-github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
+pytestmark: MarkDecorator = pytest.mark.asyncio
+github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
class TestSearchQwant(object):
@@ -14,24 +16,24 @@ class TestSearchQwant(object):
def domain() -> str:
return 'example.com'
- async def test_get_start_offset_return_0(self):
+ async def test_get_start_offset_return_0(self) -> None:
search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 0, 200)
assert search.get_start_offset() == 0
- async def test_get_start_offset_return_50(self):
+ async def test_get_start_offset_return_50(self) -> None:
search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 55, 200)
assert search.get_start_offset() == 50
- async def test_get_start_offset_return_100(self):
+ async def test_get_start_offset_return_100(self) -> None:
search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 100, 200)
assert search.get_start_offset() == 100
- async def test_get_emails(self):
+ async def test_get_emails(self) -> None:
search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 0, 200)
await search.process()
assert isinstance(await search.get_emails(), set)
- async def test_get_hostnames(self):
+ async def test_get_hostnames(self) -> None:
search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 0, 200)
await search.process()
assert isinstance(await search.get_hostnames(), list)
diff --git a/tests/discovery/test_sublist3r.py b/tests/discovery/test_sublist3r.py
index daefa121..462ee8b6 100644
--- a/tests/discovery/test_sublist3r.py
+++ b/tests/discovery/test_sublist3r.py
@@ -5,9 +5,11 @@ from theHarvester.lib.core import *
from theHarvester.discovery import sublist3r
import os
import pytest
+from _pytest.mark.structures import MarkDecorator
+from typing import Optional
-pytestmark = pytest.mark.asyncio
-github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
+pytestmark: MarkDecorator = pytest.mark.asyncio
+github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
class TestSublist3r(object):
@@ -15,14 +17,14 @@ class TestSublist3r(object):
def domain() -> str:
return 'google.com'
- async def test_api(self):
+ async def test_api(self) -> None:
base_url = f'https://api.sublist3r.com/search.php?domain={TestSublist3r.domain()}'
headers = {'User-Agent': Core.get_user_agent()}
request = requests.get(base_url, headers=headers)
assert request.status_code == 200
@pytest.mark.skipif(github_ci == 'true', reason='Skipping on Github CI due unstable site')
- async def test_do_search(self):
+ async def test_do_search(self) -> None:
search = sublist3r.SearchSublist3r(TestSublist3r.domain())
await search.process()
assert isinstance(await search.get_hostnames(), list)
diff --git a/tests/discovery/test_threatminer.py b/tests/discovery/test_threatminer.py
index e3e13f61..aff695cd 100644
--- a/tests/discovery/test_threatminer.py
+++ b/tests/discovery/test_threatminer.py
@@ -5,9 +5,11 @@ from theHarvester.lib.core import *
from theHarvester.discovery import threatminer
import os
import pytest
+from _pytest.mark.structures import MarkDecorator
+from typing import Optional
-pytestmark = pytest.mark.asyncio
-github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
+pytestmark: MarkDecorator = pytest.mark.asyncio
+github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True
class TestThreatminer(object):
@@ -15,13 +17,13 @@ class TestThreatminer(object):
def domain() -> str:
return 'target.com'
- async def test_api(self):
+ async def test_api(self) -> None:
base_url = f'https://api.threatminer.org/v2/domain.php?q={TestThreatminer.domain()}&rt=5'
headers = {'User-Agent': Core.get_user_agent()}
request = requests.get(base_url, headers=headers)
assert request.status_code == 200
- async def test_search(self):
+ async def test_search(self) -> None:
search = threatminer.SearchThreatminer(TestThreatminer.domain())
await search.process()
assert isinstance(await search.get_hostnames(), set)
diff --git a/tests/test_myparser.py b/tests/test_myparser.py
index eee0860c..6e624941 100755
--- a/tests/test_myparser.py
+++ b/tests/test_myparser.py
@@ -8,7 +8,7 @@ import pytest
class TestMyParser(object):
@pytest.mark.asyncio
- async def test_emails(self):
+ async def test_emails(self) -> None:
word = 'domain.com'
results = '@domain.com***a@domain***banotherdomain.com***c@domain.com***d@sub.domain.com***'
parse = myparser.Parser(results, word)
diff --git a/theHarvester/__main__.py b/theHarvester/__main__.py
index 91d5af99..104ae4da 100644
--- a/theHarvester/__main__.py
+++ b/theHarvester/__main__.py
@@ -1,5 +1,5 @@
#!/usr/bin/env python3
-from typing import Dict, List
+from typing import Optional, Dict, List
from theHarvester.discovery import *
from theHarvester.discovery import dnssearch, takeover, shodansearch
from theHarvester.discovery.constants import *
@@ -16,7 +16,7 @@ import string
import secrets
-async def start(rest_args=None):
+async def start(rest_args: Optional[argparse.Namespace] = None):
"""Main program function"""
parser = argparse.ArgumentParser(description='theHarvester is used to gather open source intelligence (OSINT) on a company or domain.')
parser.add_argument('-d', '--domain', help='Company name or domain to search.', required=True)
@@ -824,7 +824,7 @@ async def start(rest_args=None):
sys.exit(0)
-async def entry_point():
+async def entry_point() -> None:
try:
Core.banner()
await start()
diff --git a/theHarvester/discovery/anubis.py b/theHarvester/discovery/anubis.py
index 15cbde28..f59fb849 100644
--- a/theHarvester/discovery/anubis.py
+++ b/theHarvester/discovery/anubis.py
@@ -4,19 +4,19 @@ from theHarvester.lib.core import *
class SearchAnubis:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
- self.totalhosts = list
+ self.totalhosts: List = []
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
url = f'https://jldc.me/anubis/subdomains/{self.word}'
response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy)
- self.totalhosts: list = response[0]
+ self.totalhosts = response[0]
- async def get_hostnames(self) -> Type[list]:
+ async def get_hostnames(self) -> List:
return self.totalhosts
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/baidusearch.py b/theHarvester/discovery/baidusearch.py
index d91ad797..04c3f9cf 100644
--- a/theHarvester/discovery/baidusearch.py
+++ b/theHarvester/discovery/baidusearch.py
@@ -4,7 +4,7 @@ from theHarvester.parsers import myparser
class SearchBaidu:
- def __init__(self, word, limit):
+ def __init__(self, word, limit) -> None:
self.word = word
self.total_results = ""
self.server = 'www.baidu.com'
@@ -12,7 +12,7 @@ class SearchBaidu:
self.limit = limit
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
headers = {
'Host': self.hostname,
'User-agent': Core.get_user_agent()
@@ -23,7 +23,7 @@ class SearchBaidu:
for response in responses:
self.total_results += response
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/bevigil.py b/theHarvester/discovery/bevigil.py
index b99f0b41..0cafd074 100644
--- a/theHarvester/discovery/bevigil.py
+++ b/theHarvester/discovery/bevigil.py
@@ -1,16 +1,17 @@
from theHarvester.lib.core import *
+from typing import Set
class SearchBeVigil:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
- self.totalhosts = set()
- self.interestingurls = set()
+ self.totalhosts: Set = set()
+ self.interestingurls: Set = set()
self.key = Core.bevigil_key()
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
subdomain_endpoint = f"https://osint.bevigil.com/api/{self.word}/subdomains/"
url_endpoint = f"https://osint.bevigil.com/api/{self.word}/urls/"
headers = {'X-Access-Token': self.key}
@@ -31,6 +32,6 @@ class SearchBeVigil:
async def get_interestingurls(self) -> set:
return self.interestingurls
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/binaryedgesearch.py b/theHarvester/discovery/binaryedgesearch.py
index 8382e9c6..51a33c6e 100644
--- a/theHarvester/discovery/binaryedgesearch.py
+++ b/theHarvester/discovery/binaryedgesearch.py
@@ -1,12 +1,13 @@
from theHarvester.discovery.constants import *
+from typing import Set
import asyncio
class SearchBinaryEdge:
- def __init__(self, word, limit):
+ def __init__(self, word, limit) -> None:
self.word = word
- self.totalhosts = set()
+ self.totalhosts: Set = set()
self.proxy = False
self.key = Core.binaryedge_key()
self.limit = 501 if limit >= 501 else limit
@@ -14,7 +15,7 @@ class SearchBinaryEdge:
if self.key is None:
raise MissingKey('binaryedge')
- async def do_search(self):
+ async def do_search(self) -> None:
base_url = f'https://api.binaryedge.io/v2/query/domains/subdomain/{self.word}'
headers = {'X-KEY': self.key, 'User-Agent': Core.get_user_agent()}
for page in range(1, self.limit):
@@ -35,6 +36,6 @@ class SearchBinaryEdge:
async def get_hostnames(self) -> set:
return self.totalhosts
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/bingsearch.py b/theHarvester/discovery/bingsearch.py
index b7ff5017..8e23b8f2 100644
--- a/theHarvester/discovery/bingsearch.py
+++ b/theHarvester/discovery/bingsearch.py
@@ -5,7 +5,7 @@ from theHarvester.parsers import myparser
class SearchBing:
- def __init__(self, word, limit, start):
+ def __init__(self, word, limit, start) -> None:
self.word = word.replace(' ', '%20')
self.results = ""
self.total_results = ""
@@ -17,7 +17,7 @@ class SearchBing:
self.counter = start
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
headers = {
'Host': self.hostname,
'Cookie': 'SRCHHPGUSR=ADLT=DEMOTE&NRSLT=50',
@@ -30,7 +30,7 @@ class SearchBing:
for response in responses:
self.total_results += response
- async def do_search_api(self):
+ async def do_search_api(self) -> None:
url = 'https://api.cognitive.microsoft.com/bing/v7.0/search?'
params = {
'q': self.word,
@@ -43,7 +43,7 @@ class SearchBing:
self.results = await AsyncFetcher.fetch_all([url], headers=headers, params=params, proxy=self.proxy)
self.total_results += self.results
- async def do_search_vhost(self):
+ async def do_search_vhost(self) -> None:
headers = {
'Host': self.hostname,
'Cookie': 'mkt=en-US;ui=en-US;SRCHHPGUSR=NEWWND=0&ADLT=DEMOTE&NRSLT=50',
@@ -68,7 +68,7 @@ class SearchBing:
rawres = myparser.Parser(self.total_results, self.word)
return await rawres.hostnames_all()
- async def process(self, api, proxy=False):
+ async def process(self, api, proxy: bool=False) -> None:
self.proxy = proxy
if api == 'yes':
if self.bingApi is None:
@@ -80,5 +80,5 @@ class SearchBing:
await self.do_search()
print(f'\tSearching {self.counter} results.')
- async def process_vhost(self):
+ async def process_vhost(self) -> None:
await self.do_search_vhost()
diff --git a/theHarvester/discovery/bufferoverun.py b/theHarvester/discovery/bufferoverun.py
index f0be34d2..50f7ca4b 100644
--- a/theHarvester/discovery/bufferoverun.py
+++ b/theHarvester/discovery/bufferoverun.py
@@ -3,13 +3,13 @@ import re
class SearchBufferover:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
self.totalhosts = set()
self.totalips = set()
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
url = f'https://dns.bufferover.run/dns?q={self.word}'
responses = await AsyncFetcher.fetch_all(urls=[url], json=True, proxy=self.proxy)
responses = responses[0]
@@ -30,6 +30,6 @@ class SearchBufferover:
async def get_ips(self) -> set:
return self.totalips
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/censysearch.py b/theHarvester/discovery/censysearch.py
index cdbc4c69..13f99c9c 100644
--- a/theHarvester/discovery/censysearch.py
+++ b/theHarvester/discovery/censysearch.py
@@ -1,3 +1,4 @@
+from typing import Set
from theHarvester.discovery.constants import MissingKey
from theHarvester.lib.core import Core
from censys.search import CensysCertificates
@@ -9,17 +10,17 @@ from censys.common.exceptions import (
class SearchCensys:
- def __init__(self, domain, limit=500):
+ def __init__(self, domain, limit: int=500) -> None:
self.word = domain
self.key = Core.censys_key()
if self.key[0] is None or self.key[1] is None:
raise MissingKey("Censys ID and/or Secret")
- self.totalhosts = set()
- self.emails = set()
+ self.totalhosts: Set = set()
+ self.emails: Set = set()
self.limit = limit
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
try:
cert_search = CensysCertificates(
api_id=self.key[0],
@@ -48,6 +49,6 @@ class SearchCensys:
async def get_emails(self) -> set:
return self.emails
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/certspottersearch.py b/theHarvester/discovery/certspottersearch.py
index d00fe9a1..b4efe40d 100644
--- a/theHarvester/discovery/certspottersearch.py
+++ b/theHarvester/discovery/certspottersearch.py
@@ -1,11 +1,12 @@
from theHarvester.lib.core import *
+from typing import Set
class SearchCertspoter:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
- self.totalhosts = set()
+ self.totalhosts: Set = set()
self.proxy = False
async def do_search(self) -> None:
@@ -28,7 +29,7 @@ class SearchCertspoter:
async def get_hostnames(self) -> set:
return self.totalhosts
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
print('\tSearching results.')
diff --git a/theHarvester/discovery/constants.py b/theHarvester/discovery/constants.py
index 5090c433..4921be68 100644
--- a/theHarvester/discovery/constants.py
+++ b/theHarvester/discovery/constants.py
@@ -109,7 +109,7 @@ class MissingKey(Exception):
"""
:raise: When there is a module that has not been provided its API key
"""
- def __init__(self, source: Optional[str]):
+ def __init__(self, source: Optional[str]) -> None:
if source:
self.message = f'\n\033[93m[!] Missing API key for {source}. \033[0m'
else:
diff --git a/theHarvester/discovery/crtsh.py b/theHarvester/discovery/crtsh.py
index 4a759ddd..13f7f259 100644
--- a/theHarvester/discovery/crtsh.py
+++ b/theHarvester/discovery/crtsh.py
@@ -4,9 +4,9 @@ from typing import List, Set
class SearchCrtsh:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
- self.data = set()
+ self.data: Set = set()
self.proxy = False
async def do_search(self) -> List:
@@ -21,14 +21,14 @@ class SearchCrtsh:
data = {domain for domain in data if (domain[0] != '*' and str(domain[0:4]).isnumeric() is False)}
except Exception as e:
print(e)
- clean = []
+ clean: List = []
for x in data:
pre = x.split()
for y in pre:
clean.append(y)
return clean
- async def process(self, proxy=False) -> None:
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
data = await self.do_search()
self.data = data
diff --git a/theHarvester/discovery/dnsdumpster.py b/theHarvester/discovery/dnsdumpster.py
index fcf33ddc..257a9f0f 100644
--- a/theHarvester/discovery/dnsdumpster.py
+++ b/theHarvester/discovery/dnsdumpster.py
@@ -6,14 +6,14 @@ import asyncio
class SearchDnsDumpster:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word.replace(' ', '%20')
self.results = ""
self.totalresults = ""
self.server = 'dnsdumpster.com'
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
try:
agent = Core.get_user_agent()
headers = {'User-Agent': agent}
@@ -53,6 +53,6 @@ class SearchDnsDumpster:
rawres = myparser.Parser(self.totalresults, self.word)
return await rawres.hostnames()
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search() # Only need to do it once.
diff --git a/theHarvester/discovery/dnssearch.py b/theHarvester/discovery/dnssearch.py
index a5a4e456..09f1ef78 100644
--- a/theHarvester/discovery/dnssearch.py
+++ b/theHarvester/discovery/dnssearch.py
@@ -24,7 +24,7 @@ from theHarvester.lib import hostchecker
class DnsForce:
- def __init__(self, domain, dnsserver, verbose=False):
+ def __init__(self, domain, dnsserver, verbose: bool=False) -> None:
self.domain = domain
self.subdo = False
self.verbose = verbose
@@ -59,8 +59,8 @@ class DnsForce:
IP_REGEX = r'\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}'
PORT_REGEX = r'\d{1,5}'
-NETMASK_REGEX = r'\d{1,2}|' + IP_REGEX
-NETWORK_REGEX = r'\b({})(?:\:({}))?(?:\/({}))?\b'.format(
+NETMASK_REGEX: str = r'\d{1,2}|' + IP_REGEX
+NETWORK_REGEX: str = r'\b({})(?:\:({}))?(?:\/({}))?\b'.format(
IP_REGEX,
PORT_REGEX,
NETMASK_REGEX)
diff --git a/theHarvester/discovery/duckduckgosearch.py b/theHarvester/discovery/duckduckgosearch.py
index 3e5608eb..d6ece9c2 100644
--- a/theHarvester/discovery/duckduckgosearch.py
+++ b/theHarvester/discovery/duckduckgosearch.py
@@ -2,23 +2,24 @@ from theHarvester.discovery.constants import *
from theHarvester.lib.core import *
from theHarvester.parsers import myparser
import json
+from typing import Union
class SearchDuckDuckGo:
- def __init__(self, word, limit):
+ def __init__(self, word, limit) -> None:
self.word = word
self.results = ""
self.totalresults = ""
- self.dorks = []
- self.links = []
+ self.dorks: List = []
+ self.links: List = []
self.database = 'https://duckduckgo.com/?q='
self.api = 'https://api.duckduckgo.com/?q=x&format=json&pretty=1' # Currently using API.
self.quantity = '100'
self.limit = limit
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
# Do normal scraping.
url = self.api.replace('x', self.word)
headers = {'User-Agent': googleUA}
@@ -30,7 +31,7 @@ class SearchDuckDuckGo:
all_resps = await AsyncFetcher.fetch_all(urls)
self.totalresults += ''.join(all_resps)
- async def crawl(self, text):
+ async def crawl(self, text: Union[bytes, str]):
"""
Function parses json and returns URLs.
:param text: formatted json
@@ -80,6 +81,6 @@ class SearchDuckDuckGo:
rawres = myparser.Parser(self.totalresults, self.word)
return await rawres.hostnames()
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search() # Only need to search once since using API.
diff --git a/theHarvester/discovery/fullhuntsearch.py b/theHarvester/discovery/fullhuntsearch.py
index 5dc32c1d..911a5ac0 100644
--- a/theHarvester/discovery/fullhuntsearch.py
+++ b/theHarvester/discovery/fullhuntsearch.py
@@ -4,7 +4,7 @@ from theHarvester.lib.core import *
class SearchFullHunt:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
self.key = Core.fullhunt_key()
if self.key is None:
@@ -12,16 +12,16 @@ class SearchFullHunt:
self.total_results = None
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
url = f'https://fullhunt.io/api/v1/domain/{self.word}/subdomains'
response = await AsyncFetcher.fetch_all([url], json=True, headers={'User-Agent': Core.get_user_agent(),
'X-API-KEY': self.key},
proxy=self.proxy)
self.total_results = response[0]['hosts']
- async def get_hostnames(self) -> set:
+ async def get_hostnames(self):
return self.total_results
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/githubcode.py b/theHarvester/discovery/githubcode.py
index 14b53624..9d1511c6 100644
--- a/theHarvester/discovery/githubcode.py
+++ b/theHarvester/discovery/githubcode.py
@@ -1,7 +1,7 @@
from theHarvester.discovery.constants import *
from theHarvester.lib.core import *
from theHarvester.parsers import myparser
-from typing import List, Dict, Any, Optional, NamedTuple, Tuple
+from typing import Union, List, Dict, Any, Optional, NamedTuple, Tuple
import asyncio
import aiohttp
import urllib.parse as urlparse
@@ -25,13 +25,13 @@ class ErrorResult(NamedTuple):
class SearchGithubCode:
- def __init__(self, word, limit):
+ def __init__(self, word, limit) -> None:
self.word = word
self.total_results = ""
self.server = 'api.github.com'
self.limit = limit
- self.counter = 0
- self.page = 1
+ self.counter: int = 0
+ self.page: int = 1
self.key = Core.github_key()
# If you don't have a personal access token, github narrows your search capabilities significantly
# rate limits you more severely
@@ -63,7 +63,7 @@ class SearchGithubCode:
else:
return None
- async def handle_response(self, response: Tuple[str, dict, int, Any]):
+ async def handle_response(self, response: Tuple[str, dict, int, Any]) -> Union[ErrorResult, RetryResult, SuccessResult]:
text, json_data, status, links = response
if status == 200:
results = await self.fragments_from_response(json_data)
@@ -78,7 +78,7 @@ class SearchGithubCode:
except ValueError:
return ErrorResult(status, text)
- async def do_search(self, page: Optional[int]) -> Tuple[str, dict, int, Any]:
+ async def do_search(self, page: int) -> Tuple[str, dict, int, Any]:
if page is None:
url = f'https://{self.server}/search/code?q="{self.word}"'
else:
@@ -99,13 +99,13 @@ class SearchGithubCode:
return await resp.text(), await resp.json(), resp.status, resp.links
@staticmethod
- async def next_page_or_end(result: SuccessResult) -> Optional[int]:
+ async def next_page_or_end(result: SuccessResult) -> int:
if result.next_page is not None:
return result.next_page
else:
return result.last_page
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
try:
while self.counter <= self.limit and self.page is not None:
diff --git a/theHarvester/discovery/hackertarget.py b/theHarvester/discovery/hackertarget.py
index 224d08bf..6bd3f427 100644
--- a/theHarvester/discovery/hackertarget.py
+++ b/theHarvester/discovery/hackertarget.py
@@ -6,21 +6,21 @@ class SearchHackerTarget:
Class uses the HackerTarget api to gather subdomains and ips
"""
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
self.total_results = ""
self.hostname = 'https://api.hackertarget.com'
self.proxy = False
self.results = None
- async def do_search(self):
+ async def do_search(self) -> None:
headers = {'User-agent': Core.get_user_agent()}
urls = [f'{self.hostname}/hostsearch/?q={self.word}', f'{self.hostname}/reversedns/?q={self.word}']
responses = await AsyncFetcher.fetch_all(urls, headers=headers, proxy=self.proxy)
for response in responses:
self.total_results += response.replace(",", ":")
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/huntersearch.py b/theHarvester/discovery/huntersearch.py
index 7d32d484..ae48514e 100644
--- a/theHarvester/discovery/huntersearch.py
+++ b/theHarvester/discovery/huntersearch.py
@@ -1,10 +1,11 @@
from theHarvester.discovery.constants import *
from theHarvester.lib.core import *
+from typing import List
class SearchHunter:
- def __init__(self, word, limit, start):
+ def __init__(self, word, limit, start) -> None:
self.word = word
self.limit = limit
self.limit = 10 if limit > 10 else limit
@@ -16,10 +17,10 @@ class SearchHunter:
self.counter = start
self.database = f'https://api.hunter.io/v2/domain-search?domain={self.word}&api_key={self.key}&limit=10'
self.proxy = False
- self.hostnames = []
- self.emails = []
+ self.hostnames: List = []
+ self.emails: List = []
- async def do_search(self):
+ async def do_search(self) -> None:
# First determine if user account is not a free account, this call is free
is_free = True
headers = {'User-Agent': Core.get_user_agent()}
@@ -66,7 +67,7 @@ class SearchHunter:
if self.word in source['domain']}))
return emails, domains
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search() # Only need to do it once.
diff --git a/theHarvester/discovery/intelxsearch.py b/theHarvester/discovery/intelxsearch.py
index af32f561..a74396bc 100644
--- a/theHarvester/discovery/intelxsearch.py
+++ b/theHarvester/discovery/intelxsearch.py
@@ -8,7 +8,7 @@ import requests
class SearchIntelx:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
self.key = Core.intelx_key()
if self.key is None:
@@ -16,11 +16,11 @@ class SearchIntelx:
self.database = 'https://2.intelx.io'
self.results = None
self.info = ()
- self.limit = 10000
+ self.limit: int = 10000
self.proxy = False
self.offset = -1
- async def do_search(self):
+ async def do_search(self) -> None:
try:
# Based on: https://github.com/IntelligenceX/SDK/blob/master/Python/intelxapi.py
# API requests self identification
@@ -53,7 +53,7 @@ class SearchIntelx:
except Exception as e:
print(f'An exception has occurred in Intelx: {e}')
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False):
self.proxy = proxy
await self.do_search()
intelx_parser = intelxparser.Parser()
diff --git a/theHarvester/discovery/omnisint.py b/theHarvester/discovery/omnisint.py
index b2891ba1..f04c79b8 100644
--- a/theHarvester/discovery/omnisint.py
+++ b/theHarvester/discovery/omnisint.py
@@ -2,13 +2,13 @@ from theHarvester.lib.core import *
class SearchOmnisint:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
self.totalhosts = set()
self.totalips = set()
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
base_url = f'https://sonar.omnisint.io/all/{self.word}?page=1'
responses = await AsyncFetcher.fetch_all([base_url], json=True, headers={'User-Agent': Core.get_user_agent()},
proxy=self.proxy)
@@ -20,6 +20,6 @@ class SearchOmnisint:
async def get_ips(self) -> set:
return self.totalips
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/otxsearch.py b/theHarvester/discovery/otxsearch.py
index 9dab2419..db44ae24 100644
--- a/theHarvester/discovery/otxsearch.py
+++ b/theHarvester/discovery/otxsearch.py
@@ -1,24 +1,25 @@
+from typing import Set
from theHarvester.lib.core import *
import re
class SearchOtx:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
- self.totalhosts = set()
- self.totalips = set()
+ self.totalhosts: Set = set()
+ self.totalips: Set = set()
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
url = f'https://otx.alienvault.com/api/v1/indicators/domain/{self.word}/passive_dns'
response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy)
responses = response[0]
dct = responses
- self.totalhosts: set = {host['hostname'] for host in dct['passive_dns']}
+ self.totalhosts = {host['hostname'] for host in dct['passive_dns']}
# filter out ips that are just called NXDOMAIN
- self.totalips: set = {ip['address'] for ip in dct['passive_dns']
- if re.match(r"^\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}$", ip['address'])}
+ self.totalips = {ip['address'] for ip in dct['passive_dns'] if re.match(r"^\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}$",
+ ip['address'])}
async def get_hostnames(self) -> set:
return self.totalhosts
@@ -26,6 +27,6 @@ class SearchOtx:
async def get_ips(self) -> set:
return self.totalips
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/pentesttools.py b/theHarvester/discovery/pentesttools.py
index 5ab8b8c8..77ba8e4a 100644
--- a/theHarvester/discovery/pentesttools.py
+++ b/theHarvester/discovery/pentesttools.py
@@ -1,18 +1,19 @@
from theHarvester.discovery.constants import *
from theHarvester.lib.core import *
+from typing import List
import json
import time
class SearchPentestTools:
- def __init__(self, word):
+ def __init__(self, word) -> None:
# Script is largely based off https://pentest-tools.com/public/api_client.py.txt
self.word = word
self.key = Core.pentest_tools_key()
if self.key is None:
raise MissingKey('PentestTools')
- self.total_results = []
+ self.total_results: List = []
self.api = f'https://pentest-tools.com/api?key={self.key}'
self.proxy = False
@@ -57,7 +58,7 @@ class SearchPentestTools:
async def get_hostnames(self) -> list:
return self.total_results
- async def do_search(self):
+ async def do_search(self) -> None:
subdomain_payload = {
'op': 'start_scan',
'tool_id': 20,
@@ -73,6 +74,6 @@ class SearchPentestTools:
scan_id = res_json['scan_id']
await self.poll(scan_id)
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search() # Only need to do it once.
diff --git a/theHarvester/discovery/projectdiscovery.py b/theHarvester/discovery/projectdiscovery.py
index 5a730b6f..a2a40542 100644
--- a/theHarvester/discovery/projectdiscovery.py
+++ b/theHarvester/discovery/projectdiscovery.py
@@ -4,7 +4,7 @@ from theHarvester.lib.core import *
class SearchDiscovery:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
self.key = Core.projectdiscovery_key()
if self.key is None:
@@ -19,9 +19,9 @@ class SearchDiscovery:
proxy=self.proxy)
self.total_results = [f'{domains}.{self.word}' for domains in response[0]['subdomains']]
- async def get_hostnames(self) -> set:
+ async def get_hostnames(self):
return self.total_results
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/qwantsearch.py b/theHarvester/discovery/qwantsearch.py
index 81ec8198..1c174c50 100644
--- a/theHarvester/discovery/qwantsearch.py
+++ b/theHarvester/discovery/qwantsearch.py
@@ -7,7 +7,7 @@ from theHarvester.parsers import myparser
class SearchQwant:
- def __init__(self, word, start, limit):
+ def __init__(self, word, start, limit) -> None:
self.word = word
self.total_results = ""
self.limit = int(limit)
@@ -78,6 +78,6 @@ class SearchQwant:
parser = myparser.Parser(self.total_results, self.word)
return await parser.hostnames()
- async def process(self, proxy=False) -> None:
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/rapiddns.py b/theHarvester/discovery/rapiddns.py
index 799c6b26..bb670cd8 100644
--- a/theHarvester/discovery/rapiddns.py
+++ b/theHarvester/discovery/rapiddns.py
@@ -4,9 +4,9 @@ from theHarvester.lib.core import *
class SearchRapidDns:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
- self.total_results = []
+ self.total_results: List = []
self.proxy = False
async def do_search(self):
@@ -35,7 +35,7 @@ class SearchRapidDns:
except Exception as e:
print(f'An exception has occurred: {str(e)}')
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/rocketreach.py b/theHarvester/discovery/rocketreach.py
index 82457669..03144b63 100644
--- a/theHarvester/discovery/rocketreach.py
+++ b/theHarvester/discovery/rocketreach.py
@@ -1,23 +1,24 @@
from theHarvester.discovery.constants import *
from theHarvester.lib.core import *
+from typing import Set
import asyncio
class SearchRocketReach:
- def __init__(self, word, limit):
- self.ips = set()
+ def __init__(self, word, limit) -> None:
+ self.ips: Set = set()
self.word = word
self.key = Core.rocketreach_key()
if self.key is None:
raise MissingKey('RocketReach')
- self.hosts = set()
+ self.hosts: Set = set()
self.proxy = False
self.baseurl = 'https://api.rocketreach.co/v2/api/search'
- self.links = set()
+ self.links: Set = set()
self.limit = limit
- async def do_search(self):
+ async def do_search(self) -> None:
try:
headers = {
'Api-Key': self.key,
@@ -56,6 +57,6 @@ class SearchRocketReach:
async def get_links(self):
return self.links
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/securitytrailssearch.py b/theHarvester/discovery/securitytrailssearch.py
index f0b9b071..6560181b 100644
--- a/theHarvester/discovery/securitytrailssearch.py
+++ b/theHarvester/discovery/securitytrailssearch.py
@@ -6,7 +6,7 @@ import asyncio
class SearchSecuritytrail:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
self.key = Core.security_trails_key()
if self.key is None:
@@ -41,7 +41,7 @@ class SearchSecuritytrail:
self.results = subdomain_response[0]
self.totalresults += self.results
- async def process(self, proxy=False) -> None:
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.authenticate()
await self.do_search()
diff --git a/theHarvester/discovery/shodansearch.py b/theHarvester/discovery/shodansearch.py
index 90db6f20..bf8e81ec 100644
--- a/theHarvester/discovery/shodansearch.py
+++ b/theHarvester/discovery/shodansearch.py
@@ -7,12 +7,12 @@ from collections import OrderedDict
class SearchShodan:
- def __init__(self):
+ def __init__(self) -> None:
self.key = Core.shodan_key()
if self.key is None:
raise MissingKey('Shodan')
self.api = Shodan(self.key)
- self.hostdatarow = []
+ self.hostdatarow: List = []
self.tracker: OrderedDict = OrderedDict()
async def search_ip(self, ip):
@@ -20,16 +20,16 @@ class SearchShodan:
ipaddress = ip
results = self.api.host(ipaddress)
asn = ''
- domains = list()
- hostnames = list()
+ domains: List = list()
+ hostnames: List = list()
ip_str = ''
isp = ''
org = ''
- ports = list()
+ ports: List = list()
title = ''
server = ''
product = ''
- technologies = list()
+ technologies: List = list()
data_first_dict = dict(results['data'][0])
diff --git a/theHarvester/discovery/sublist3r.py b/theHarvester/discovery/sublist3r.py
index ee0611ae..b3465e90 100644
--- a/theHarvester/discovery/sublist3r.py
+++ b/theHarvester/discovery/sublist3r.py
@@ -4,12 +4,12 @@ from theHarvester.lib.core import *
class SearchSublist3r:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
self.totalhosts = list
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
url = f'https://api.sublist3r.com/search.php?domain={self.word}'
response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy)
self.totalhosts: list = response[0]
@@ -17,6 +17,6 @@ class SearchSublist3r:
async def get_hostnames(self) -> Type[list]:
return self.totalhosts
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/takeover.py b/theHarvester/discovery/takeover.py
index 1f0b1a1a..f7085457 100644
--- a/theHarvester/discovery/takeover.py
+++ b/theHarvester/discovery/takeover.py
@@ -4,7 +4,7 @@ import re
class TakeOver:
- def __init__(self, hosts):
+ def __init__(self, hosts) -> None:
# NOTE THIS MODULE IS ACTIVE RECON
self.hosts = hosts
self.results = ""
@@ -34,7 +34,7 @@ class TakeOver:
'page not found': 'Uptimerobot',
'project not found': 'Surge.sh'}
- async def check(self, url, resp):
+ async def check(self, url, resp) -> None:
# Simple function that takes response and checks if any fingerprints exists
# If a fingerprint exists figures out which one and prints it out
regex = re.compile("(?=(" + "|".join(map(re.escape, list(self.fingerprints.keys()))) + "))")
@@ -46,7 +46,7 @@ class TakeOver:
# Sanity check as to not error out
print(f'\t\033[91m Type of takeover is: {self.fingerprints[match]}\033[1;32;40m')
- async def do_take(self):
+ async def do_take(self) -> None:
try:
if len(self.hosts) > 0:
tup_resps: list = await AsyncFetcher.fetch_all(self.hosts, takeover=True, proxy=self.proxy)
@@ -60,6 +60,6 @@ class TakeOver:
except Exception as e:
print(e)
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_take()
diff --git a/theHarvester/discovery/threatcrowd.py b/theHarvester/discovery/threatcrowd.py
index 78cbbfc3..d0404785 100644
--- a/theHarvester/discovery/threatcrowd.py
+++ b/theHarvester/discovery/threatcrowd.py
@@ -4,13 +4,13 @@ from theHarvester.lib.core import *
class SearchThreatcrowd:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word.replace(' ', '%20')
- self.hostnames = list()
- self.ips = list()
+ self.hostnames: List = list()
+ self.ips: List = list()
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
base_url = f'https://www.threatcrowd.org/searchApi/v2/domain/report/?domain={self.word}'
headers = {'User-Agent': Core.get_user_agent()}
try:
@@ -27,7 +27,7 @@ class SearchThreatcrowd:
async def get_hostnames(self) -> List:
return self.hostnames
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
await self.get_hostnames()
diff --git a/theHarvester/discovery/threatminer.py b/theHarvester/discovery/threatminer.py
index 251cb89b..3357c6c5 100644
--- a/theHarvester/discovery/threatminer.py
+++ b/theHarvester/discovery/threatminer.py
@@ -1,32 +1,32 @@
-from typing import Type
+from typing import Type, List
from theHarvester.lib.core import *
class SearchThreatminer:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
- self.totalhosts = list
- self.totalips = list
+ self.totalhosts: List = []
+ self.totalips: List = []
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
url = f'https://api.threatminer.org/v2/domain.php?q={self.word}&rt=5'
response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy)
- self.totalhosts: set = {host for host in response[0]['results']}
+ self.totalhosts = {host for host in response[0]['results']}
second_url = f'https://api.threatminer.org/v2/domain.php?q={self.word}&rt=2'
secondresp = await AsyncFetcher.fetch_all([second_url], json=True, proxy=self.proxy)
try:
- self.totalips: set = {resp['ip'] for resp in secondresp[0]['results']}
+ self.totalips = {resp['ip'] for resp in secondresp[0]['results']}
except TypeError:
pass
- async def get_hostnames(self) -> Type[list]:
+ async def get_hostnames(self) -> list:
return self.totalhosts
- async def get_ips(self) -> Type[list]:
+ async def get_ips(self) -> list:
return self.totalips
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/urlscan.py b/theHarvester/discovery/urlscan.py
index c2a1371f..0bdb6dc2 100644
--- a/theHarvester/discovery/urlscan.py
+++ b/theHarvester/discovery/urlscan.py
@@ -3,15 +3,15 @@ from theHarvester.lib.core import *
class SearchUrlscan:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.word = word
- self.totalhosts = list()
- self.totalips = list()
- self.interestingurls = list()
- self.totalasns = list()
+ self.totalhosts: List = list()
+ self.totalips: List = list()
+ self.interestingurls: List = list()
+ self.totalasns: List = list()
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
url = f'https://urlscan.io/api/v1/search/?q=domain:{self.word}'
response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy)
resp = response[0]
@@ -32,6 +32,6 @@ class SearchUrlscan:
async def get_asns(self) -> List:
return self.totalasns
- async def process(self, proxy=False):
+ async def process(self, proxy: bool = False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/virustotal.py b/theHarvester/discovery/virustotal.py
index 13bb72dd..33559db1 100644
--- a/theHarvester/discovery/virustotal.py
+++ b/theHarvester/discovery/virustotal.py
@@ -4,15 +4,15 @@ from theHarvester.lib.core import *
class SearchVirustotal:
- def __init__(self, word):
+ def __init__(self, word) -> None:
self.key = Core.virustotal_key()
if self.key is None:
raise MissingKey('virustotal')
self.word = word
self.proxy = False
- self.hostnames = []
+ self.hostnames: List = []
- async def do_search(self):
+ async def do_search(self) -> None:
# TODO determine if more endpoints can yield useful info given a domain
# based on: https://developers.virustotal.com/reference/domains-relationships
# base_url = "https://www.virustotal.com/api/v3/domains/domain/subdomains?limit=40"
@@ -81,6 +81,6 @@ class SearchVirustotal:
total_subdomains.sort()
return total_subdomains
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search()
diff --git a/theHarvester/discovery/yahoosearch.py b/theHarvester/discovery/yahoosearch.py
index 6e94c3b4..92b81609 100644
--- a/theHarvester/discovery/yahoosearch.py
+++ b/theHarvester/discovery/yahoosearch.py
@@ -4,14 +4,14 @@ from theHarvester.parsers import myparser
class SearchYahoo:
- def __init__(self, word, limit):
+ def __init__(self, word, limit) -> None:
self.word = word
self.total_results = ""
self.server = 'search.yahoo.com'
self.limit = limit
self.proxy = False
- async def do_search(self):
+ async def do_search(self) -> None:
base_url = f'https://{self.server}/search?p=%40{self.word}&b=xx&pz=10'
headers = {
'Host': self.server,
@@ -22,7 +22,7 @@ class SearchYahoo:
for response in responses:
self.total_results += response
- async def process(self):
+ async def process(self) -> None:
await self.do_search()
async def get_emails(self):
@@ -38,7 +38,7 @@ class SearchYahoo:
emails.add(email)
return list(emails)
- async def get_hostnames(self, proxy=False):
+ async def get_hostnames(self, proxy: bool=False):
self.proxy = proxy
rawres = myparser.Parser(self.total_results, self.word)
return await rawres.hostnames()
diff --git a/theHarvester/discovery/zoomeyesearch.py b/theHarvester/discovery/zoomeyesearch.py
index 3d230b5f..d04a9a20 100644
--- a/theHarvester/discovery/zoomeyesearch.py
+++ b/theHarvester/discovery/zoomeyesearch.py
@@ -1,13 +1,14 @@
from theHarvester.discovery.constants import *
from theHarvester.lib.core import *
from theHarvester.parsers import myparser
+from typing import List
import asyncio
import re
class SearchZoomEye:
- def __init__(self, word, limit):
+ def __init__(self, word, limit) -> None:
self.word = word
self.limit = limit
self.key = Core.zoomeye_key()
@@ -19,11 +20,11 @@ class SearchZoomEye:
raise MissingKey('zoomeye')
self.baseurl = 'https://api.zoomeye.org/host/search'
self.proxy = False
- self.totalasns = list()
- self.totalhosts = list()
- self.interestingurls = list()
- self.totalips = list()
- self.totalemails = list()
+ self.totalasns: List = list()
+ self.totalhosts: List = list()
+ self.interestingurls: List = list()
+ self.totalips: List = list()
+ self.totalemails: List = list()
# Regex used is directly from: https://github.com/GerbenJavado/LinkFinder/blob/master/linkfinder.py#L29
# Maybe one day it will be a pip package
# Regardless LinkFinder is an amazing tool!
@@ -56,7 +57,7 @@ class SearchZoomEye:
"""
self.iurl_regex = re.compile(self.iurl_regex, re.VERBOSE)
- async def fetch_subdomains(self):
+ async def fetch_subdomains(self) -> None:
# Based on docs from: https://www.zoomeye.org/doc#search-sub-domain-ip
headers = {
'API-KEY': self.key,
@@ -91,7 +92,7 @@ class SearchZoomEye:
if i % 10 == 0:
await asyncio.sleep(get_delay() + 1)
- async def do_search(self):
+ async def do_search(self) -> None:
headers = {
'API-KEY': self.key,
'User-Agent': Core.get_user_agent()
@@ -211,7 +212,7 @@ class SearchZoomEye:
print(f'An exception has occurred: {e}')
return hostnames, emails, ips, asns, iurls
- async def process(self, proxy=False):
+ async def process(self, proxy: bool=False) -> None:
self.proxy = proxy
await self.do_search() # Only need to do it once.
diff --git a/theHarvester/lib/api/api.py b/theHarvester/lib/api/api.py
index 548022af..0849896c 100644
--- a/theHarvester/lib/api/api.py
+++ b/theHarvester/lib/api/api.py
@@ -1,5 +1,5 @@
import argparse
-from typing import List
+from typing import Any, Dict, Union, List
import os
from fastapi import FastAPI, Header, Query, Request
from fastapi.responses import HTMLResponse, UJSONResponse
@@ -27,7 +27,7 @@ except RuntimeError:
@app.get('/', response_class=HTMLResponse)
-async def root(*, user_agent: str = Header(None)):
+async def root(*, user_agent: str = Header(None)) -> Union[RedirectResponse, str]:
# very basic user agent filtering
if user_agent and ('gobuster' in user_agent or 'sqlmap' in user_agent or 'rustbuster' in user_agent):
response = RedirectResponse(app.url_path_for('bot'))
@@ -59,7 +59,7 @@ async def root(*, user_agent: str = Header(None)):
@app.get('/nicebot')
-async def bot():
+async def bot() -> Dict[str, str]:
# nice bot
string = {'bot': 'These are not the droids you are looking for'}
return string
@@ -77,7 +77,7 @@ async def getsources(request: Request):
@app.get('/dnsbrute', response_class=UJSONResponse)
@limiter.limit('5/minute')
async def dnsbrute(request: Request, user_agent: str = Header(None),
- domain: str = Query(..., description='Domain to be brute forced')):
+ domain: str = Query(..., description='Domain to be brute forced')) -> Union[Dict[str, Any], RedirectResponse]:
# Endpoint for user to signal to do DNS brute forcing
# Rate limit of 5 requests per minute
# basic user agent filtering
@@ -112,7 +112,7 @@ async def query(request: Request, dns_server: str = Query(""), user_agent: str =
take_over: bool = Query(False), virtual_host: bool = Query(False),
source: List[str] = Query(..., description='Data sources to query comma separated with no space'),
limit: int = Query(500), start: int = Query(0),
- domain: str = Query(..., description='Domain to be harvested')):
+ domain: str = Query(..., description='Domain to be harvested')) -> Union[Dict[str, Any], RedirectResponse]:
# Query function that allows user to query theHarvester rest API
# Rate limit of 2 requests per minute
diff --git a/theHarvester/lib/api/api_example.py b/theHarvester/lib/api/api_example.py
index 1c4761ae..e9799d77 100644
--- a/theHarvester/lib/api/api_example.py
+++ b/theHarvester/lib/api/api_example.py
@@ -17,7 +17,7 @@ async def fetch(session, url):
return await response.text()
-async def main():
+async def main() -> None:
"""
Just a simple example of how to interact with the rest api
you can easily use requests instead of aiohttp or whatever you best see fit
diff --git a/theHarvester/lib/core.py b/theHarvester/lib/core.py
index 2d35e02b..d5efaafb 100644
--- a/theHarvester/lib/core.py
+++ b/theHarvester/lib/core.py
@@ -1,6 +1,6 @@
# coding=utf-8
from __future__ import annotations
-from typing import Union, Any, Tuple, List
+from typing import Sized, Union, Any, Tuple, List
import yaml
import asyncio
import aiohttp
@@ -238,7 +238,7 @@ class AsyncFetcher:
proxy_list = Core.proxy_list()
@classmethod
- async def post_fetch(cls, url, headers='', data='', params='', json=False, proxy=False):
+ async def post_fetch(cls, url, headers: Sized='', data: str='', params: str='', json: bool=False, proxy: bool=False):
if len(headers) == 0:
headers = {'User-Agent': Core.get_user_agent()}
timeout = aiohttp.ClientTimeout(total=720)
@@ -273,7 +273,7 @@ class AsyncFetcher:
return ''
@staticmethod
- async def fetch(session, url, params='', json=False, proxy="") -> Union[str, dict, list, bool]:
+ async def fetch(session, url, params: str = '', json: bool = False, proxy: str = "") -> Union[str, dict, list, bool]:
# This fetch method solely focuses on get requests
try:
# Wrap in try except due to 0x89 png/jpg files
@@ -306,7 +306,7 @@ class AsyncFetcher:
return ''
@staticmethod
- async def takeover_fetch(session, url, proxy="") -> Union[Tuple[Any, Any], str]:
+ async def takeover_fetch(session, url: str, proxy: str = "") -> Union[Tuple[Any, Any], str]:
# This fetch method solely focuses on get requests
try:
# Wrap in try except due to 0x89 png/jpg files
@@ -326,7 +326,8 @@ class AsyncFetcher:
return url, ''
@classmethod
- async def fetch_all(cls, urls, headers='', params='', json=False, takeover=False, proxy=False) -> tuple:
+ async def fetch_all(cls, urls, headers: Sized='', params: Sized='', json: bool = False, takeover: bool = False,
+ proxy: bool = False) -> tuple:
# By default, timeout is 5 minutes; 60 seconds should suffice
timeout = aiohttp.ClientTimeout(total=60)
if len(headers) == 0:
diff --git a/theHarvester/lib/hostchecker.py b/theHarvester/lib/hostchecker.py
index b8127eb5..82287c4e 100644
--- a/theHarvester/lib/hostchecker.py
+++ b/theHarvester/lib/hostchecker.py
@@ -8,16 +8,16 @@ Revised to use aiodns & asyncio on 2019-09-23
import aiodns
import asyncio
import socket
-from typing import Tuple, Any
+from typing import Tuple, Any, List, Set
class Checker:
- def __init__(self, hosts: list, nameserver=False):
+ def __init__(self, hosts: list, nameserver: bool = False) -> None:
self.hosts = hosts
- self.realhosts: list = []
- self.addresses: set = set()
- self.nameserver = []
+ self.realhosts: List = []
+ self.addresses: Set = set()
+ self.nameserver: List = []
if nameserver:
self.nameserver = nameserver
@@ -33,7 +33,8 @@ class Checker:
except Exception:
return f"{host}", tuple()
- async def query_all(self, resolver) -> list:
+ async def query_all(self, resolver) -> tuple[
+ BaseException | Any, BaseException | Any, BaseException | Any, BaseException | Any, BaseException | Any]:
results = await asyncio.gather(*[asyncio.create_task(self.query(host, resolver))
for host in self.hosts])
return results
diff --git a/theHarvester/lib/stash.py b/theHarvester/lib/stash.py
index 2a8ed66d..4a6ebb6f 100644
--- a/theHarvester/lib/stash.py
+++ b/theHarvester/lib/stash.py
@@ -1,6 +1,8 @@
import aiosqlite
import datetime
import os
+from sqlite3.dbapi2 import Row
+from typing import Iterable, Optional, Union, List, Dict
db_path = os.path.expanduser('~/.local/share/theHarvester')
@@ -10,24 +12,24 @@ if not os.path.isdir(db_path):
class StashManager:
- def __init__(self):
+ def __init__(self) -> None:
self.db = os.path.join(db_path, 'stash.sqlite')
self.results = ""
self.totalresults = ""
- self.latestscandomain = {}
- self.domainscanhistory = []
- self.scanboarddata = {}
- self.scanstats = []
- self.latestscanresults = []
- self.previousscanresults = []
+ self.latestscandomain: Dict = {}
+ self.domainscanhistory: List = []
+ self.scanboarddata: Dict = {}
+ self.scanstats: List = []
+ self.latestscanresults: List = []
+ self.previousscanresults: List = []
- async def do_init(self):
+ async def do_init(self) -> None:
async with aiosqlite.connect(self.db) as db:
await db.execute(
'CREATE TABLE IF NOT EXISTS results (domain text, resource text, type text, find_date date, source text)')
await db.commit()
- async def store(self, domain, resource, res_type, source):
+ async def store(self, domain, resource, res_type, source) -> None:
self.domain = domain
self.resource = resource
self.type = res_type
@@ -41,7 +43,7 @@ class StashManager:
except Exception as e:
print(e)
- async def store_all(self, domain, all, res_type, source):
+ async def store_all(self, domain, all, res_type, source) -> None:
self.domain = domain
self.all = all
self.type = res_type
@@ -109,7 +111,7 @@ class StashManager:
except Exception as e:
print(e)
- async def getlatestscanresults(self, domain, previousday=False):
+ async def getlatestscanresults(self, domain, previousday: bool=False) -> Optional[Iterable[Union[Row, str]]]:
try:
async with aiosqlite.connect(self.db, timeout=30) as conn:
if previousday:
@@ -217,7 +219,7 @@ class StashManager:
except Exception as e:
print(e)
- async def getpluginscanstatistics(self):
+ async def getpluginscanstatistics(self) -> Optional[Iterable[Row]]:
try:
async with aiosqlite.connect(self.db, timeout=30) as conn:
cursor = await conn.execute('''
diff --git a/theHarvester/parsers/intelxparser.py b/theHarvester/parsers/intelxparser.py
index aa56a42a..d61de739 100644
--- a/theHarvester/parsers/intelxparser.py
+++ b/theHarvester/parsers/intelxparser.py
@@ -1,8 +1,11 @@
+from typing import Set
+
+
class Parser:
- def __init__(self):
- self.emails = set()
- self.hosts = set()
+ def __init__(self) -> None:
+ self.emails: Set = set()
+ self.hosts: Set = set()
async def parse_dictionaries(self, results: dict) -> tuple:
"""
diff --git a/theHarvester/parsers/myparser.py b/theHarvester/parsers/myparser.py
index 7a8eb4ac..7b2bb0ae 100644
--- a/theHarvester/parsers/myparser.py
+++ b/theHarvester/parsers/myparser.py
@@ -1,14 +1,15 @@
import re
+from typing import Set, List
class Parser:
- def __init__(self, results, word):
+ def __init__(self, results, word) -> None:
self.results = results
self.word = word
- self.temp = []
+ self.temp: List = []
- async def genericClean(self):
+ async def genericClean(self) -> None:
self.results = self.results.replace('', '').replace('', '').replace('', '').replace('', '') \
.replace('%3a', '').replace('', '').replace('', '') \
.replace('', '').replace('', '')
@@ -16,7 +17,7 @@ class Parser:
for search in ('<', '>', ':', '=', ';', '&', '%3A', '%3D', '%3C', '%2f', '/', '\\'):
self.results = self.results.replace(search, ' ')
- async def urlClean(self):
+ async def urlClean(self) -> None:
self.results = self.results.replace('', '').replace('', '').replace('%2f', '').replace('%3a', '')
for search in ('<', '>', ':', '=', ';', '&', '%3A', '%3D', '%3C'):
self.results = self.results.replace(search, ' ')
@@ -34,7 +35,7 @@ class Parser:
return true_emails
async def fileurls(self, file):
- urls = []
+ urls: List = []
reg_urls = re.compile(' Set[str]:
found = re.finditer(r'(http|https)://(www\.)?trello.com/([a-zA-Z\d\-_\.]+/?)*', self.results)
urls = {match.group().strip() for match in found}
return urls
diff --git a/theHarvester/parsers/securitytrailsparser.py b/theHarvester/parsers/securitytrailsparser.py
index 85ffcde5..bfda60b3 100644
--- a/theHarvester/parsers/securitytrailsparser.py
+++ b/theHarvester/parsers/securitytrailsparser.py
@@ -1,13 +1,13 @@
-from typing import Union, Tuple, List
+from typing import Union, Tuple, List, Set
class Parser:
- def __init__(self, word, text):
+ def __init__(self, word, text) -> None:
self.word = word
self.text = text
- self.hostnames = set()
- self.ips = set()
+ self.hostnames: Set = set()
+ self.ips: Set = set()
async def parse_text(self) -> Union[List, Tuple]:
sub_domain_flag = 0
diff --git a/theHarvester/screenshot/screenshot.py b/theHarvester/screenshot/screenshot.py
index 575f9c07..0e7425ab 100644
--- a/theHarvester/screenshot/screenshot.py
+++ b/theHarvester/screenshot/screenshot.py
@@ -11,16 +11,17 @@ from datetime import datetime
import os
import ssl
import sys
+from typing import Sized, Tuple
class ScreenShotter:
- def __init__(self, output):
+ def __init__(self, output) -> None:
self.output = output
self.slash = "\\" if 'win' in sys.platform else '/'
self.slash = "" if (self.output[-1] == "\\" or self.output[-1] == "/") else self.slash
- def verify_path(self):
+ def verify_path(self) -> bool:
try:
if not os.path.isdir(self.output):
answer = input(
@@ -36,19 +37,19 @@ class ScreenShotter:
return False
@staticmethod
- async def verify_installation():
+ async def verify_installation() -> None:
# Helper function that verifies pyppeteer & chromium are installed
# If chromium is not installed pyppeteer will prompt user to install it
browser = await launch(headless=True, ignoreHTTPSErrors=True, args=["--no-sandbox"])
await browser.close()
@staticmethod
- def chunk_list(items, chunk_size):
+ def chunk_list(items: Sized, chunk_size):
# Based off of: https://github.com/apache/incubator-sdap-ingester
return [items[i:i + chunk_size] for i in range(0, len(items), chunk_size)]
@staticmethod
- async def visit(url):
+ async def visit(url: str) -> Tuple[str, str]:
try:
# print(f'attempting to visit: {url}')
timeout = aiohttp.ClientTimeout(total=35)
@@ -67,7 +68,7 @@ class ScreenShotter:
print(f'An exception has occurred while attempting to visit {url} : {e}')
return "", ""
- async def take_screenshot(self, url):
+ async def take_screenshot(self, url: str) -> Tuple[str, ...]:
url = f'http://{url}' if not url.startswith('http') else url
url = url.replace('www.', '')
print(f'Attempting to take a screenshot of: {url}')