diff --git a/restfulHarvest.py b/restfulHarvest.py index 7cc31812..d7b02cf7 100755 --- a/restfulHarvest.py +++ b/restfulHarvest.py @@ -8,7 +8,7 @@ parser.add_argument('-p', '--port', default=5000, help='Port to bind the web ser parser.add_argument('-l', '--log-level', default='info', help='Set logging level, default is info but [critical|error|warning|info|debug|trace] can be set') parser.add_argument('-r', '--reload', default=False, help='Enable automatic reload used during development of the api', action='store_true') -args = parser.parse_args() +args: argparse.Namespace = parser.parse_args() if __name__ == '__main__': uvicorn.run('theHarvester.lib.api.api:app', host=args.host, port=args.port, log_level=args.log_level, reload=args.reload) diff --git a/setup.py b/setup.py index d09dea9c..dd01fe6c 100755 --- a/setup.py +++ b/setup.py @@ -2,7 +2,7 @@ from setuptools import setup, find_packages from theHarvester.lib.core import Core with open('README.md', 'r') as fh: - long_description = fh.read() + long_description: str = fh.read() setup( name='theHarvester', @@ -20,9 +20,8 @@ setup( classifiers=[ "Programming Language :: Python :: 3", - "Programming Language :: Python :: 3.8", - "Programming Language :: Python :: 3.9", "Programming Language :: Python :: 3.10", + "Programming Language :: Python :: 3.11", "License :: OSI Approved :: GNU General Public License v2 (GPLv2)", "Operating System :: OS Independent", ], diff --git a/tests/discovery/test_anubis.py b/tests/discovery/test_anubis.py index 684ae85c..1b1d492b 100644 --- a/tests/discovery/test_anubis.py +++ b/tests/discovery/test_anubis.py @@ -5,9 +5,11 @@ from theHarvester.lib.core import * from theHarvester.discovery import anubis import os import pytest +from _pytest.mark.structures import MarkDecorator +from typing import Optional -pytestmark = pytest.mark.asyncio -github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True +pytestmark: MarkDecorator = pytest.mark.asyncio +github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True class TestAnubis: @@ -15,7 +17,7 @@ class TestAnubis: def domain() -> str: return 'apple.com' - async def test_api(self): + async def test_api(self) -> None: base_url = f'https://jldc.me/anubis/subdomains/{TestAnubis.domain()}' headers = {'User-Agent': Core.get_user_agent()} request = requests.get(base_url, headers=headers) @@ -26,6 +28,6 @@ class TestAnubis: await search.do_search() return await search.get_hostnames() - async def test_process(self): + async def test_process(self) -> None: await self.test_do_search() assert len(await self.test_do_search()) > 0 diff --git a/tests/discovery/test_certspotter.py b/tests/discovery/test_certspotter.py index aa73e39b..19152693 100644 --- a/tests/discovery/test_certspotter.py +++ b/tests/discovery/test_certspotter.py @@ -5,9 +5,11 @@ from theHarvester.discovery import certspottersearch import os import requests import pytest +from _pytest.mark.structures import MarkDecorator +from typing import Optional -pytestmark = pytest.mark.asyncio -github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True +pytestmark: MarkDecorator = pytest.mark.asyncio +github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True class TestCertspotter(object): @@ -15,18 +17,18 @@ class TestCertspotter(object): def domain() -> str: return 'metasploit.com' - async def test_api(self): + async def test_api(self) -> None: base_url = f'https://api.certspotter.com/v1/issuances?domain={TestCertspotter.domain()}&expand=dns_names' headers = {'User-Agent': Core.get_user_agent()} request = requests.get(base_url, headers=headers) assert request.status_code == 200 - async def test_search(self): + async def test_search(self) -> None: search = certspottersearch.SearchCertspoter(TestCertspotter.domain()) await search.process() assert isinstance(await search.get_hostnames(), set) - async def test_search_no_results(self): + async def test_search_no_results(self) -> None: search = certspottersearch.SearchCertspoter('radiant.eu') await search.process() assert len(await search.get_hostnames()) == 0 diff --git a/tests/discovery/test_githubcode.py b/tests/discovery/test_githubcode.py index 0e7d52d6..0eb9154d 100644 --- a/tests/discovery/test_githubcode.py +++ b/tests/discovery/test_githubcode.py @@ -4,8 +4,9 @@ from theHarvester.lib.core import Core from unittest.mock import MagicMock from requests import Response import pytest +from _pytest.mark.structures import MarkDecorator -pytestmark = pytest.mark.asyncio +pytestmark: MarkDecorator = pytest.mark.asyncio class TestSearchGithubCode: @@ -65,31 +66,31 @@ class TestSearchGithubCode: response.json = MagicMock(return_value=json) response.status_code = 200 - async def test_missing_key(self): + async def test_missing_key(self) -> None: with pytest.raises(MissingKey): Core.github_key = MagicMock(return_value=None) githubcode.SearchGithubCode(word="test", limit=500) - async def test_fragments_from_response(self): + async def test_fragments_from_response(self) -> None: Core.github_key = MagicMock(return_value="lol") test_class_instance = githubcode.SearchGithubCode(word="test", limit=500) test_result = await test_class_instance.fragments_from_response(self.OkResponse.response.json()) print('test_result: ', test_result) assert test_result == ["test1", "test2"] - async def test_invalid_fragments_from_response(self): + async def test_invalid_fragments_from_response(self) -> None: Core.github_key = MagicMock(return_value="lol") test_class_instance = githubcode.SearchGithubCode(word="test", limit=500) test_result = await test_class_instance.fragments_from_response(self.MalformedResponse.response.json()) assert test_result == [] - async def test_next_page(self): + async def test_next_page(self) -> None: Core.github_key = MagicMock(return_value="lol") test_class_instance = githubcode.SearchGithubCode(word="test", limit=500) test_result = githubcode.SuccessResult(list(), next_page=2, last_page=4) assert (2 == await test_class_instance.next_page_or_end(test_result)) - async def test_last_page(self): + async def test_last_page(self) -> None: Core.github_key = MagicMock(return_value="lol") test_class_instance = githubcode.SearchGithubCode(word="test", limit=500) test_result = githubcode.SuccessResult(list(), None, None) diff --git a/tests/discovery/test_omnisint.py b/tests/discovery/test_omnisint.py index bc492ee0..3277be2c 100644 --- a/tests/discovery/test_omnisint.py +++ b/tests/discovery/test_omnisint.py @@ -5,9 +5,11 @@ from theHarvester.discovery import omnisint import os import requests import pytest +from _pytest.mark.structures import MarkDecorator +from typing import Optional -pytestmark = pytest.mark.asyncio -github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True +pytestmark: MarkDecorator = pytest.mark.asyncio +github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True class TestOmnisint(object): @@ -16,13 +18,13 @@ class TestOmnisint(object): return 'uber.com' @pytest.mark.skipif(github_ci == 'true', reason='Skipping on Github CI due to unstable status code from site') - async def test_api(self): + async def test_api(self) -> None: base_url = f'https://sonar.omnisint.io/all/{TestOmnisint.domain()}' headers = {'User-Agent': Core.get_user_agent()} request = requests.get(base_url, headers=headers) assert request.status_code == 200 - async def test_search(self): + async def test_search(self) -> None: search = omnisint.SearchOmnisint(TestOmnisint.domain()) await search.process() assert isinstance(await search.get_hostnames(), list) diff --git a/tests/discovery/test_otx.py b/tests/discovery/test_otx.py index acc41c4c..851afb9d 100644 --- a/tests/discovery/test_otx.py +++ b/tests/discovery/test_otx.py @@ -5,9 +5,11 @@ from theHarvester.discovery import otxsearch import os import requests import pytest +from _pytest.mark.structures import MarkDecorator +from typing import Optional -pytestmark = pytest.mark.asyncio -github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True +pytestmark: MarkDecorator = pytest.mark.asyncio +github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True class TestOtx(object): @@ -15,13 +17,13 @@ class TestOtx(object): def domain() -> str: return 'metasploit.com' - async def test_api(self): + async def test_api(self) -> None: base_url = f'https://otx.alienvault.com/api/v1/indicators/domain/{TestOtx.domain()}/passive_dns' headers = {'User-Agent': Core.get_user_agent()} request = requests.get(base_url, headers=headers) assert request.status_code == 200 - async def test_search(self): + async def test_search(self) -> None: search = otxsearch.SearchOtx(TestOtx.domain()) await search.process() assert isinstance(await search.get_hostnames(), set) diff --git a/tests/discovery/test_qwantsearch.py b/tests/discovery/test_qwantsearch.py index 2fdcad4d..7ce21f75 100644 --- a/tests/discovery/test_qwantsearch.py +++ b/tests/discovery/test_qwantsearch.py @@ -3,9 +3,11 @@ from theHarvester.discovery import qwantsearch import os import pytest +from _pytest.mark.structures import MarkDecorator +from typing import Optional -pytestmark = pytest.mark.asyncio -github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True +pytestmark: MarkDecorator = pytest.mark.asyncio +github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True class TestSearchQwant(object): @@ -14,24 +16,24 @@ class TestSearchQwant(object): def domain() -> str: return 'example.com' - async def test_get_start_offset_return_0(self): + async def test_get_start_offset_return_0(self) -> None: search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 0, 200) assert search.get_start_offset() == 0 - async def test_get_start_offset_return_50(self): + async def test_get_start_offset_return_50(self) -> None: search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 55, 200) assert search.get_start_offset() == 50 - async def test_get_start_offset_return_100(self): + async def test_get_start_offset_return_100(self) -> None: search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 100, 200) assert search.get_start_offset() == 100 - async def test_get_emails(self): + async def test_get_emails(self) -> None: search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 0, 200) await search.process() assert isinstance(await search.get_emails(), set) - async def test_get_hostnames(self): + async def test_get_hostnames(self) -> None: search = qwantsearch.SearchQwant(TestSearchQwant.domain(), 0, 200) await search.process() assert isinstance(await search.get_hostnames(), list) diff --git a/tests/discovery/test_sublist3r.py b/tests/discovery/test_sublist3r.py index daefa121..462ee8b6 100644 --- a/tests/discovery/test_sublist3r.py +++ b/tests/discovery/test_sublist3r.py @@ -5,9 +5,11 @@ from theHarvester.lib.core import * from theHarvester.discovery import sublist3r import os import pytest +from _pytest.mark.structures import MarkDecorator +from typing import Optional -pytestmark = pytest.mark.asyncio -github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True +pytestmark: MarkDecorator = pytest.mark.asyncio +github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True class TestSublist3r(object): @@ -15,14 +17,14 @@ class TestSublist3r(object): def domain() -> str: return 'google.com' - async def test_api(self): + async def test_api(self) -> None: base_url = f'https://api.sublist3r.com/search.php?domain={TestSublist3r.domain()}' headers = {'User-Agent': Core.get_user_agent()} request = requests.get(base_url, headers=headers) assert request.status_code == 200 @pytest.mark.skipif(github_ci == 'true', reason='Skipping on Github CI due unstable site') - async def test_do_search(self): + async def test_do_search(self) -> None: search = sublist3r.SearchSublist3r(TestSublist3r.domain()) await search.process() assert isinstance(await search.get_hostnames(), list) diff --git a/tests/discovery/test_threatminer.py b/tests/discovery/test_threatminer.py index e3e13f61..aff695cd 100644 --- a/tests/discovery/test_threatminer.py +++ b/tests/discovery/test_threatminer.py @@ -5,9 +5,11 @@ from theHarvester.lib.core import * from theHarvester.discovery import threatminer import os import pytest +from _pytest.mark.structures import MarkDecorator +from typing import Optional -pytestmark = pytest.mark.asyncio -github_ci = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True +pytestmark: MarkDecorator = pytest.mark.asyncio +github_ci: Optional[str] = os.getenv('GITHUB_ACTIONS') # Github set this to be the following: true instead of True class TestThreatminer(object): @@ -15,13 +17,13 @@ class TestThreatminer(object): def domain() -> str: return 'target.com' - async def test_api(self): + async def test_api(self) -> None: base_url = f'https://api.threatminer.org/v2/domain.php?q={TestThreatminer.domain()}&rt=5' headers = {'User-Agent': Core.get_user_agent()} request = requests.get(base_url, headers=headers) assert request.status_code == 200 - async def test_search(self): + async def test_search(self) -> None: search = threatminer.SearchThreatminer(TestThreatminer.domain()) await search.process() assert isinstance(await search.get_hostnames(), set) diff --git a/tests/test_myparser.py b/tests/test_myparser.py index eee0860c..6e624941 100755 --- a/tests/test_myparser.py +++ b/tests/test_myparser.py @@ -8,7 +8,7 @@ import pytest class TestMyParser(object): @pytest.mark.asyncio - async def test_emails(self): + async def test_emails(self) -> None: word = 'domain.com' results = '@domain.com***a@domain***banotherdomain.com***c@domain.com***d@sub.domain.com***' parse = myparser.Parser(results, word) diff --git a/theHarvester/__main__.py b/theHarvester/__main__.py index 91d5af99..104ae4da 100644 --- a/theHarvester/__main__.py +++ b/theHarvester/__main__.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -from typing import Dict, List +from typing import Optional, Dict, List from theHarvester.discovery import * from theHarvester.discovery import dnssearch, takeover, shodansearch from theHarvester.discovery.constants import * @@ -16,7 +16,7 @@ import string import secrets -async def start(rest_args=None): +async def start(rest_args: Optional[argparse.Namespace] = None): """Main program function""" parser = argparse.ArgumentParser(description='theHarvester is used to gather open source intelligence (OSINT) on a company or domain.') parser.add_argument('-d', '--domain', help='Company name or domain to search.', required=True) @@ -824,7 +824,7 @@ async def start(rest_args=None): sys.exit(0) -async def entry_point(): +async def entry_point() -> None: try: Core.banner() await start() diff --git a/theHarvester/discovery/anubis.py b/theHarvester/discovery/anubis.py index 15cbde28..f59fb849 100644 --- a/theHarvester/discovery/anubis.py +++ b/theHarvester/discovery/anubis.py @@ -4,19 +4,19 @@ from theHarvester.lib.core import * class SearchAnubis: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word - self.totalhosts = list + self.totalhosts: List = [] self.proxy = False - async def do_search(self): + async def do_search(self) -> None: url = f'https://jldc.me/anubis/subdomains/{self.word}' response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy) - self.totalhosts: list = response[0] + self.totalhosts = response[0] - async def get_hostnames(self) -> Type[list]: + async def get_hostnames(self) -> List: return self.totalhosts - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/baidusearch.py b/theHarvester/discovery/baidusearch.py index d91ad797..04c3f9cf 100644 --- a/theHarvester/discovery/baidusearch.py +++ b/theHarvester/discovery/baidusearch.py @@ -4,7 +4,7 @@ from theHarvester.parsers import myparser class SearchBaidu: - def __init__(self, word, limit): + def __init__(self, word, limit) -> None: self.word = word self.total_results = "" self.server = 'www.baidu.com' @@ -12,7 +12,7 @@ class SearchBaidu: self.limit = limit self.proxy = False - async def do_search(self): + async def do_search(self) -> None: headers = { 'Host': self.hostname, 'User-agent': Core.get_user_agent() @@ -23,7 +23,7 @@ class SearchBaidu: for response in responses: self.total_results += response - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/bevigil.py b/theHarvester/discovery/bevigil.py index b99f0b41..0cafd074 100644 --- a/theHarvester/discovery/bevigil.py +++ b/theHarvester/discovery/bevigil.py @@ -1,16 +1,17 @@ from theHarvester.lib.core import * +from typing import Set class SearchBeVigil: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word - self.totalhosts = set() - self.interestingurls = set() + self.totalhosts: Set = set() + self.interestingurls: Set = set() self.key = Core.bevigil_key() self.proxy = False - async def do_search(self): + async def do_search(self) -> None: subdomain_endpoint = f"https://osint.bevigil.com/api/{self.word}/subdomains/" url_endpoint = f"https://osint.bevigil.com/api/{self.word}/urls/" headers = {'X-Access-Token': self.key} @@ -31,6 +32,6 @@ class SearchBeVigil: async def get_interestingurls(self) -> set: return self.interestingurls - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/binaryedgesearch.py b/theHarvester/discovery/binaryedgesearch.py index 8382e9c6..51a33c6e 100644 --- a/theHarvester/discovery/binaryedgesearch.py +++ b/theHarvester/discovery/binaryedgesearch.py @@ -1,12 +1,13 @@ from theHarvester.discovery.constants import * +from typing import Set import asyncio class SearchBinaryEdge: - def __init__(self, word, limit): + def __init__(self, word, limit) -> None: self.word = word - self.totalhosts = set() + self.totalhosts: Set = set() self.proxy = False self.key = Core.binaryedge_key() self.limit = 501 if limit >= 501 else limit @@ -14,7 +15,7 @@ class SearchBinaryEdge: if self.key is None: raise MissingKey('binaryedge') - async def do_search(self): + async def do_search(self) -> None: base_url = f'https://api.binaryedge.io/v2/query/domains/subdomain/{self.word}' headers = {'X-KEY': self.key, 'User-Agent': Core.get_user_agent()} for page in range(1, self.limit): @@ -35,6 +36,6 @@ class SearchBinaryEdge: async def get_hostnames(self) -> set: return self.totalhosts - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/bingsearch.py b/theHarvester/discovery/bingsearch.py index b7ff5017..8e23b8f2 100644 --- a/theHarvester/discovery/bingsearch.py +++ b/theHarvester/discovery/bingsearch.py @@ -5,7 +5,7 @@ from theHarvester.parsers import myparser class SearchBing: - def __init__(self, word, limit, start): + def __init__(self, word, limit, start) -> None: self.word = word.replace(' ', '%20') self.results = "" self.total_results = "" @@ -17,7 +17,7 @@ class SearchBing: self.counter = start self.proxy = False - async def do_search(self): + async def do_search(self) -> None: headers = { 'Host': self.hostname, 'Cookie': 'SRCHHPGUSR=ADLT=DEMOTE&NRSLT=50', @@ -30,7 +30,7 @@ class SearchBing: for response in responses: self.total_results += response - async def do_search_api(self): + async def do_search_api(self) -> None: url = 'https://api.cognitive.microsoft.com/bing/v7.0/search?' params = { 'q': self.word, @@ -43,7 +43,7 @@ class SearchBing: self.results = await AsyncFetcher.fetch_all([url], headers=headers, params=params, proxy=self.proxy) self.total_results += self.results - async def do_search_vhost(self): + async def do_search_vhost(self) -> None: headers = { 'Host': self.hostname, 'Cookie': 'mkt=en-US;ui=en-US;SRCHHPGUSR=NEWWND=0&ADLT=DEMOTE&NRSLT=50', @@ -68,7 +68,7 @@ class SearchBing: rawres = myparser.Parser(self.total_results, self.word) return await rawres.hostnames_all() - async def process(self, api, proxy=False): + async def process(self, api, proxy: bool=False) -> None: self.proxy = proxy if api == 'yes': if self.bingApi is None: @@ -80,5 +80,5 @@ class SearchBing: await self.do_search() print(f'\tSearching {self.counter} results.') - async def process_vhost(self): + async def process_vhost(self) -> None: await self.do_search_vhost() diff --git a/theHarvester/discovery/bufferoverun.py b/theHarvester/discovery/bufferoverun.py index f0be34d2..50f7ca4b 100644 --- a/theHarvester/discovery/bufferoverun.py +++ b/theHarvester/discovery/bufferoverun.py @@ -3,13 +3,13 @@ import re class SearchBufferover: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word self.totalhosts = set() self.totalips = set() self.proxy = False - async def do_search(self): + async def do_search(self) -> None: url = f'https://dns.bufferover.run/dns?q={self.word}' responses = await AsyncFetcher.fetch_all(urls=[url], json=True, proxy=self.proxy) responses = responses[0] @@ -30,6 +30,6 @@ class SearchBufferover: async def get_ips(self) -> set: return self.totalips - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/censysearch.py b/theHarvester/discovery/censysearch.py index cdbc4c69..13f99c9c 100644 --- a/theHarvester/discovery/censysearch.py +++ b/theHarvester/discovery/censysearch.py @@ -1,3 +1,4 @@ +from typing import Set from theHarvester.discovery.constants import MissingKey from theHarvester.lib.core import Core from censys.search import CensysCertificates @@ -9,17 +10,17 @@ from censys.common.exceptions import ( class SearchCensys: - def __init__(self, domain, limit=500): + def __init__(self, domain, limit: int=500) -> None: self.word = domain self.key = Core.censys_key() if self.key[0] is None or self.key[1] is None: raise MissingKey("Censys ID and/or Secret") - self.totalhosts = set() - self.emails = set() + self.totalhosts: Set = set() + self.emails: Set = set() self.limit = limit self.proxy = False - async def do_search(self): + async def do_search(self) -> None: try: cert_search = CensysCertificates( api_id=self.key[0], @@ -48,6 +49,6 @@ class SearchCensys: async def get_emails(self) -> set: return self.emails - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/certspottersearch.py b/theHarvester/discovery/certspottersearch.py index d00fe9a1..b4efe40d 100644 --- a/theHarvester/discovery/certspottersearch.py +++ b/theHarvester/discovery/certspottersearch.py @@ -1,11 +1,12 @@ from theHarvester.lib.core import * +from typing import Set class SearchCertspoter: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word - self.totalhosts = set() + self.totalhosts: Set = set() self.proxy = False async def do_search(self) -> None: @@ -28,7 +29,7 @@ class SearchCertspoter: async def get_hostnames(self) -> set: return self.totalhosts - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() print('\tSearching results.') diff --git a/theHarvester/discovery/constants.py b/theHarvester/discovery/constants.py index 5090c433..4921be68 100644 --- a/theHarvester/discovery/constants.py +++ b/theHarvester/discovery/constants.py @@ -109,7 +109,7 @@ class MissingKey(Exception): """ :raise: When there is a module that has not been provided its API key """ - def __init__(self, source: Optional[str]): + def __init__(self, source: Optional[str]) -> None: if source: self.message = f'\n\033[93m[!] Missing API key for {source}. \033[0m' else: diff --git a/theHarvester/discovery/crtsh.py b/theHarvester/discovery/crtsh.py index 4a759ddd..13f7f259 100644 --- a/theHarvester/discovery/crtsh.py +++ b/theHarvester/discovery/crtsh.py @@ -4,9 +4,9 @@ from typing import List, Set class SearchCrtsh: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word - self.data = set() + self.data: Set = set() self.proxy = False async def do_search(self) -> List: @@ -21,14 +21,14 @@ class SearchCrtsh: data = {domain for domain in data if (domain[0] != '*' and str(domain[0:4]).isnumeric() is False)} except Exception as e: print(e) - clean = [] + clean: List = [] for x in data: pre = x.split() for y in pre: clean.append(y) return clean - async def process(self, proxy=False) -> None: + async def process(self, proxy: bool=False) -> None: self.proxy = proxy data = await self.do_search() self.data = data diff --git a/theHarvester/discovery/dnsdumpster.py b/theHarvester/discovery/dnsdumpster.py index fcf33ddc..257a9f0f 100644 --- a/theHarvester/discovery/dnsdumpster.py +++ b/theHarvester/discovery/dnsdumpster.py @@ -6,14 +6,14 @@ import asyncio class SearchDnsDumpster: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word.replace(' ', '%20') self.results = "" self.totalresults = "" self.server = 'dnsdumpster.com' self.proxy = False - async def do_search(self): + async def do_search(self) -> None: try: agent = Core.get_user_agent() headers = {'User-Agent': agent} @@ -53,6 +53,6 @@ class SearchDnsDumpster: rawres = myparser.Parser(self.totalresults, self.word) return await rawres.hostnames() - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() # Only need to do it once. diff --git a/theHarvester/discovery/dnssearch.py b/theHarvester/discovery/dnssearch.py index a5a4e456..09f1ef78 100644 --- a/theHarvester/discovery/dnssearch.py +++ b/theHarvester/discovery/dnssearch.py @@ -24,7 +24,7 @@ from theHarvester.lib import hostchecker class DnsForce: - def __init__(self, domain, dnsserver, verbose=False): + def __init__(self, domain, dnsserver, verbose: bool=False) -> None: self.domain = domain self.subdo = False self.verbose = verbose @@ -59,8 +59,8 @@ class DnsForce: IP_REGEX = r'\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}' PORT_REGEX = r'\d{1,5}' -NETMASK_REGEX = r'\d{1,2}|' + IP_REGEX -NETWORK_REGEX = r'\b({})(?:\:({}))?(?:\/({}))?\b'.format( +NETMASK_REGEX: str = r'\d{1,2}|' + IP_REGEX +NETWORK_REGEX: str = r'\b({})(?:\:({}))?(?:\/({}))?\b'.format( IP_REGEX, PORT_REGEX, NETMASK_REGEX) diff --git a/theHarvester/discovery/duckduckgosearch.py b/theHarvester/discovery/duckduckgosearch.py index 3e5608eb..d6ece9c2 100644 --- a/theHarvester/discovery/duckduckgosearch.py +++ b/theHarvester/discovery/duckduckgosearch.py @@ -2,23 +2,24 @@ from theHarvester.discovery.constants import * from theHarvester.lib.core import * from theHarvester.parsers import myparser import json +from typing import Union class SearchDuckDuckGo: - def __init__(self, word, limit): + def __init__(self, word, limit) -> None: self.word = word self.results = "" self.totalresults = "" - self.dorks = [] - self.links = [] + self.dorks: List = [] + self.links: List = [] self.database = 'https://duckduckgo.com/?q=' self.api = 'https://api.duckduckgo.com/?q=x&format=json&pretty=1' # Currently using API. self.quantity = '100' self.limit = limit self.proxy = False - async def do_search(self): + async def do_search(self) -> None: # Do normal scraping. url = self.api.replace('x', self.word) headers = {'User-Agent': googleUA} @@ -30,7 +31,7 @@ class SearchDuckDuckGo: all_resps = await AsyncFetcher.fetch_all(urls) self.totalresults += ''.join(all_resps) - async def crawl(self, text): + async def crawl(self, text: Union[bytes, str]): """ Function parses json and returns URLs. :param text: formatted json @@ -80,6 +81,6 @@ class SearchDuckDuckGo: rawres = myparser.Parser(self.totalresults, self.word) return await rawres.hostnames() - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() # Only need to search once since using API. diff --git a/theHarvester/discovery/fullhuntsearch.py b/theHarvester/discovery/fullhuntsearch.py index 5dc32c1d..911a5ac0 100644 --- a/theHarvester/discovery/fullhuntsearch.py +++ b/theHarvester/discovery/fullhuntsearch.py @@ -4,7 +4,7 @@ from theHarvester.lib.core import * class SearchFullHunt: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word self.key = Core.fullhunt_key() if self.key is None: @@ -12,16 +12,16 @@ class SearchFullHunt: self.total_results = None self.proxy = False - async def do_search(self): + async def do_search(self) -> None: url = f'https://fullhunt.io/api/v1/domain/{self.word}/subdomains' response = await AsyncFetcher.fetch_all([url], json=True, headers={'User-Agent': Core.get_user_agent(), 'X-API-KEY': self.key}, proxy=self.proxy) self.total_results = response[0]['hosts'] - async def get_hostnames(self) -> set: + async def get_hostnames(self): return self.total_results - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/githubcode.py b/theHarvester/discovery/githubcode.py index 14b53624..9d1511c6 100644 --- a/theHarvester/discovery/githubcode.py +++ b/theHarvester/discovery/githubcode.py @@ -1,7 +1,7 @@ from theHarvester.discovery.constants import * from theHarvester.lib.core import * from theHarvester.parsers import myparser -from typing import List, Dict, Any, Optional, NamedTuple, Tuple +from typing import Union, List, Dict, Any, Optional, NamedTuple, Tuple import asyncio import aiohttp import urllib.parse as urlparse @@ -25,13 +25,13 @@ class ErrorResult(NamedTuple): class SearchGithubCode: - def __init__(self, word, limit): + def __init__(self, word, limit) -> None: self.word = word self.total_results = "" self.server = 'api.github.com' self.limit = limit - self.counter = 0 - self.page = 1 + self.counter: int = 0 + self.page: int = 1 self.key = Core.github_key() # If you don't have a personal access token, github narrows your search capabilities significantly # rate limits you more severely @@ -63,7 +63,7 @@ class SearchGithubCode: else: return None - async def handle_response(self, response: Tuple[str, dict, int, Any]): + async def handle_response(self, response: Tuple[str, dict, int, Any]) -> Union[ErrorResult, RetryResult, SuccessResult]: text, json_data, status, links = response if status == 200: results = await self.fragments_from_response(json_data) @@ -78,7 +78,7 @@ class SearchGithubCode: except ValueError: return ErrorResult(status, text) - async def do_search(self, page: Optional[int]) -> Tuple[str, dict, int, Any]: + async def do_search(self, page: int) -> Tuple[str, dict, int, Any]: if page is None: url = f'https://{self.server}/search/code?q="{self.word}"' else: @@ -99,13 +99,13 @@ class SearchGithubCode: return await resp.text(), await resp.json(), resp.status, resp.links @staticmethod - async def next_page_or_end(result: SuccessResult) -> Optional[int]: + async def next_page_or_end(result: SuccessResult) -> int: if result.next_page is not None: return result.next_page else: return result.last_page - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy try: while self.counter <= self.limit and self.page is not None: diff --git a/theHarvester/discovery/hackertarget.py b/theHarvester/discovery/hackertarget.py index 224d08bf..6bd3f427 100644 --- a/theHarvester/discovery/hackertarget.py +++ b/theHarvester/discovery/hackertarget.py @@ -6,21 +6,21 @@ class SearchHackerTarget: Class uses the HackerTarget api to gather subdomains and ips """ - def __init__(self, word): + def __init__(self, word) -> None: self.word = word self.total_results = "" self.hostname = 'https://api.hackertarget.com' self.proxy = False self.results = None - async def do_search(self): + async def do_search(self) -> None: headers = {'User-agent': Core.get_user_agent()} urls = [f'{self.hostname}/hostsearch/?q={self.word}', f'{self.hostname}/reversedns/?q={self.word}'] responses = await AsyncFetcher.fetch_all(urls, headers=headers, proxy=self.proxy) for response in responses: self.total_results += response.replace(",", ":") - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/huntersearch.py b/theHarvester/discovery/huntersearch.py index 7d32d484..ae48514e 100644 --- a/theHarvester/discovery/huntersearch.py +++ b/theHarvester/discovery/huntersearch.py @@ -1,10 +1,11 @@ from theHarvester.discovery.constants import * from theHarvester.lib.core import * +from typing import List class SearchHunter: - def __init__(self, word, limit, start): + def __init__(self, word, limit, start) -> None: self.word = word self.limit = limit self.limit = 10 if limit > 10 else limit @@ -16,10 +17,10 @@ class SearchHunter: self.counter = start self.database = f'https://api.hunter.io/v2/domain-search?domain={self.word}&api_key={self.key}&limit=10' self.proxy = False - self.hostnames = [] - self.emails = [] + self.hostnames: List = [] + self.emails: List = [] - async def do_search(self): + async def do_search(self) -> None: # First determine if user account is not a free account, this call is free is_free = True headers = {'User-Agent': Core.get_user_agent()} @@ -66,7 +67,7 @@ class SearchHunter: if self.word in source['domain']})) return emails, domains - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() # Only need to do it once. diff --git a/theHarvester/discovery/intelxsearch.py b/theHarvester/discovery/intelxsearch.py index af32f561..a74396bc 100644 --- a/theHarvester/discovery/intelxsearch.py +++ b/theHarvester/discovery/intelxsearch.py @@ -8,7 +8,7 @@ import requests class SearchIntelx: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word self.key = Core.intelx_key() if self.key is None: @@ -16,11 +16,11 @@ class SearchIntelx: self.database = 'https://2.intelx.io' self.results = None self.info = () - self.limit = 10000 + self.limit: int = 10000 self.proxy = False self.offset = -1 - async def do_search(self): + async def do_search(self) -> None: try: # Based on: https://github.com/IntelligenceX/SDK/blob/master/Python/intelxapi.py # API requests self identification @@ -53,7 +53,7 @@ class SearchIntelx: except Exception as e: print(f'An exception has occurred in Intelx: {e}') - async def process(self, proxy=False): + async def process(self, proxy: bool = False): self.proxy = proxy await self.do_search() intelx_parser = intelxparser.Parser() diff --git a/theHarvester/discovery/omnisint.py b/theHarvester/discovery/omnisint.py index b2891ba1..f04c79b8 100644 --- a/theHarvester/discovery/omnisint.py +++ b/theHarvester/discovery/omnisint.py @@ -2,13 +2,13 @@ from theHarvester.lib.core import * class SearchOmnisint: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word self.totalhosts = set() self.totalips = set() self.proxy = False - async def do_search(self): + async def do_search(self) -> None: base_url = f'https://sonar.omnisint.io/all/{self.word}?page=1' responses = await AsyncFetcher.fetch_all([base_url], json=True, headers={'User-Agent': Core.get_user_agent()}, proxy=self.proxy) @@ -20,6 +20,6 @@ class SearchOmnisint: async def get_ips(self) -> set: return self.totalips - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/otxsearch.py b/theHarvester/discovery/otxsearch.py index 9dab2419..db44ae24 100644 --- a/theHarvester/discovery/otxsearch.py +++ b/theHarvester/discovery/otxsearch.py @@ -1,24 +1,25 @@ +from typing import Set from theHarvester.lib.core import * import re class SearchOtx: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word - self.totalhosts = set() - self.totalips = set() + self.totalhosts: Set = set() + self.totalips: Set = set() self.proxy = False - async def do_search(self): + async def do_search(self) -> None: url = f'https://otx.alienvault.com/api/v1/indicators/domain/{self.word}/passive_dns' response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy) responses = response[0] dct = responses - self.totalhosts: set = {host['hostname'] for host in dct['passive_dns']} + self.totalhosts = {host['hostname'] for host in dct['passive_dns']} # filter out ips that are just called NXDOMAIN - self.totalips: set = {ip['address'] for ip in dct['passive_dns'] - if re.match(r"^\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}$", ip['address'])} + self.totalips = {ip['address'] for ip in dct['passive_dns'] if re.match(r"^\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}$", + ip['address'])} async def get_hostnames(self) -> set: return self.totalhosts @@ -26,6 +27,6 @@ class SearchOtx: async def get_ips(self) -> set: return self.totalips - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/pentesttools.py b/theHarvester/discovery/pentesttools.py index 5ab8b8c8..77ba8e4a 100644 --- a/theHarvester/discovery/pentesttools.py +++ b/theHarvester/discovery/pentesttools.py @@ -1,18 +1,19 @@ from theHarvester.discovery.constants import * from theHarvester.lib.core import * +from typing import List import json import time class SearchPentestTools: - def __init__(self, word): + def __init__(self, word) -> None: # Script is largely based off https://pentest-tools.com/public/api_client.py.txt self.word = word self.key = Core.pentest_tools_key() if self.key is None: raise MissingKey('PentestTools') - self.total_results = [] + self.total_results: List = [] self.api = f'https://pentest-tools.com/api?key={self.key}' self.proxy = False @@ -57,7 +58,7 @@ class SearchPentestTools: async def get_hostnames(self) -> list: return self.total_results - async def do_search(self): + async def do_search(self) -> None: subdomain_payload = { 'op': 'start_scan', 'tool_id': 20, @@ -73,6 +74,6 @@ class SearchPentestTools: scan_id = res_json['scan_id'] await self.poll(scan_id) - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() # Only need to do it once. diff --git a/theHarvester/discovery/projectdiscovery.py b/theHarvester/discovery/projectdiscovery.py index 5a730b6f..a2a40542 100644 --- a/theHarvester/discovery/projectdiscovery.py +++ b/theHarvester/discovery/projectdiscovery.py @@ -4,7 +4,7 @@ from theHarvester.lib.core import * class SearchDiscovery: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word self.key = Core.projectdiscovery_key() if self.key is None: @@ -19,9 +19,9 @@ class SearchDiscovery: proxy=self.proxy) self.total_results = [f'{domains}.{self.word}' for domains in response[0]['subdomains']] - async def get_hostnames(self) -> set: + async def get_hostnames(self): return self.total_results - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/qwantsearch.py b/theHarvester/discovery/qwantsearch.py index 81ec8198..1c174c50 100644 --- a/theHarvester/discovery/qwantsearch.py +++ b/theHarvester/discovery/qwantsearch.py @@ -7,7 +7,7 @@ from theHarvester.parsers import myparser class SearchQwant: - def __init__(self, word, start, limit): + def __init__(self, word, start, limit) -> None: self.word = word self.total_results = "" self.limit = int(limit) @@ -78,6 +78,6 @@ class SearchQwant: parser = myparser.Parser(self.total_results, self.word) return await parser.hostnames() - async def process(self, proxy=False) -> None: + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/rapiddns.py b/theHarvester/discovery/rapiddns.py index 799c6b26..bb670cd8 100644 --- a/theHarvester/discovery/rapiddns.py +++ b/theHarvester/discovery/rapiddns.py @@ -4,9 +4,9 @@ from theHarvester.lib.core import * class SearchRapidDns: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word - self.total_results = [] + self.total_results: List = [] self.proxy = False async def do_search(self): @@ -35,7 +35,7 @@ class SearchRapidDns: except Exception as e: print(f'An exception has occurred: {str(e)}') - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/rocketreach.py b/theHarvester/discovery/rocketreach.py index 82457669..03144b63 100644 --- a/theHarvester/discovery/rocketreach.py +++ b/theHarvester/discovery/rocketreach.py @@ -1,23 +1,24 @@ from theHarvester.discovery.constants import * from theHarvester.lib.core import * +from typing import Set import asyncio class SearchRocketReach: - def __init__(self, word, limit): - self.ips = set() + def __init__(self, word, limit) -> None: + self.ips: Set = set() self.word = word self.key = Core.rocketreach_key() if self.key is None: raise MissingKey('RocketReach') - self.hosts = set() + self.hosts: Set = set() self.proxy = False self.baseurl = 'https://api.rocketreach.co/v2/api/search' - self.links = set() + self.links: Set = set() self.limit = limit - async def do_search(self): + async def do_search(self) -> None: try: headers = { 'Api-Key': self.key, @@ -56,6 +57,6 @@ class SearchRocketReach: async def get_links(self): return self.links - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/securitytrailssearch.py b/theHarvester/discovery/securitytrailssearch.py index f0b9b071..6560181b 100644 --- a/theHarvester/discovery/securitytrailssearch.py +++ b/theHarvester/discovery/securitytrailssearch.py @@ -6,7 +6,7 @@ import asyncio class SearchSecuritytrail: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word self.key = Core.security_trails_key() if self.key is None: @@ -41,7 +41,7 @@ class SearchSecuritytrail: self.results = subdomain_response[0] self.totalresults += self.results - async def process(self, proxy=False) -> None: + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.authenticate() await self.do_search() diff --git a/theHarvester/discovery/shodansearch.py b/theHarvester/discovery/shodansearch.py index 90db6f20..bf8e81ec 100644 --- a/theHarvester/discovery/shodansearch.py +++ b/theHarvester/discovery/shodansearch.py @@ -7,12 +7,12 @@ from collections import OrderedDict class SearchShodan: - def __init__(self): + def __init__(self) -> None: self.key = Core.shodan_key() if self.key is None: raise MissingKey('Shodan') self.api = Shodan(self.key) - self.hostdatarow = [] + self.hostdatarow: List = [] self.tracker: OrderedDict = OrderedDict() async def search_ip(self, ip): @@ -20,16 +20,16 @@ class SearchShodan: ipaddress = ip results = self.api.host(ipaddress) asn = '' - domains = list() - hostnames = list() + domains: List = list() + hostnames: List = list() ip_str = '' isp = '' org = '' - ports = list() + ports: List = list() title = '' server = '' product = '' - technologies = list() + technologies: List = list() data_first_dict = dict(results['data'][0]) diff --git a/theHarvester/discovery/sublist3r.py b/theHarvester/discovery/sublist3r.py index ee0611ae..b3465e90 100644 --- a/theHarvester/discovery/sublist3r.py +++ b/theHarvester/discovery/sublist3r.py @@ -4,12 +4,12 @@ from theHarvester.lib.core import * class SearchSublist3r: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word self.totalhosts = list self.proxy = False - async def do_search(self): + async def do_search(self) -> None: url = f'https://api.sublist3r.com/search.php?domain={self.word}' response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy) self.totalhosts: list = response[0] @@ -17,6 +17,6 @@ class SearchSublist3r: async def get_hostnames(self) -> Type[list]: return self.totalhosts - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/takeover.py b/theHarvester/discovery/takeover.py index 1f0b1a1a..f7085457 100644 --- a/theHarvester/discovery/takeover.py +++ b/theHarvester/discovery/takeover.py @@ -4,7 +4,7 @@ import re class TakeOver: - def __init__(self, hosts): + def __init__(self, hosts) -> None: # NOTE THIS MODULE IS ACTIVE RECON self.hosts = hosts self.results = "" @@ -34,7 +34,7 @@ class TakeOver: 'page not found': 'Uptimerobot', 'project not found': 'Surge.sh'} - async def check(self, url, resp): + async def check(self, url, resp) -> None: # Simple function that takes response and checks if any fingerprints exists # If a fingerprint exists figures out which one and prints it out regex = re.compile("(?=(" + "|".join(map(re.escape, list(self.fingerprints.keys()))) + "))") @@ -46,7 +46,7 @@ class TakeOver: # Sanity check as to not error out print(f'\t\033[91m Type of takeover is: {self.fingerprints[match]}\033[1;32;40m') - async def do_take(self): + async def do_take(self) -> None: try: if len(self.hosts) > 0: tup_resps: list = await AsyncFetcher.fetch_all(self.hosts, takeover=True, proxy=self.proxy) @@ -60,6 +60,6 @@ class TakeOver: except Exception as e: print(e) - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_take() diff --git a/theHarvester/discovery/threatcrowd.py b/theHarvester/discovery/threatcrowd.py index 78cbbfc3..d0404785 100644 --- a/theHarvester/discovery/threatcrowd.py +++ b/theHarvester/discovery/threatcrowd.py @@ -4,13 +4,13 @@ from theHarvester.lib.core import * class SearchThreatcrowd: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word.replace(' ', '%20') - self.hostnames = list() - self.ips = list() + self.hostnames: List = list() + self.ips: List = list() self.proxy = False - async def do_search(self): + async def do_search(self) -> None: base_url = f'https://www.threatcrowd.org/searchApi/v2/domain/report/?domain={self.word}' headers = {'User-Agent': Core.get_user_agent()} try: @@ -27,7 +27,7 @@ class SearchThreatcrowd: async def get_hostnames(self) -> List: return self.hostnames - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() await self.get_hostnames() diff --git a/theHarvester/discovery/threatminer.py b/theHarvester/discovery/threatminer.py index 251cb89b..3357c6c5 100644 --- a/theHarvester/discovery/threatminer.py +++ b/theHarvester/discovery/threatminer.py @@ -1,32 +1,32 @@ -from typing import Type +from typing import Type, List from theHarvester.lib.core import * class SearchThreatminer: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word - self.totalhosts = list - self.totalips = list + self.totalhosts: List = [] + self.totalips: List = [] self.proxy = False - async def do_search(self): + async def do_search(self) -> None: url = f'https://api.threatminer.org/v2/domain.php?q={self.word}&rt=5' response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy) - self.totalhosts: set = {host for host in response[0]['results']} + self.totalhosts = {host for host in response[0]['results']} second_url = f'https://api.threatminer.org/v2/domain.php?q={self.word}&rt=2' secondresp = await AsyncFetcher.fetch_all([second_url], json=True, proxy=self.proxy) try: - self.totalips: set = {resp['ip'] for resp in secondresp[0]['results']} + self.totalips = {resp['ip'] for resp in secondresp[0]['results']} except TypeError: pass - async def get_hostnames(self) -> Type[list]: + async def get_hostnames(self) -> list: return self.totalhosts - async def get_ips(self) -> Type[list]: + async def get_ips(self) -> list: return self.totalips - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/urlscan.py b/theHarvester/discovery/urlscan.py index c2a1371f..0bdb6dc2 100644 --- a/theHarvester/discovery/urlscan.py +++ b/theHarvester/discovery/urlscan.py @@ -3,15 +3,15 @@ from theHarvester.lib.core import * class SearchUrlscan: - def __init__(self, word): + def __init__(self, word) -> None: self.word = word - self.totalhosts = list() - self.totalips = list() - self.interestingurls = list() - self.totalasns = list() + self.totalhosts: List = list() + self.totalips: List = list() + self.interestingurls: List = list() + self.totalasns: List = list() self.proxy = False - async def do_search(self): + async def do_search(self) -> None: url = f'https://urlscan.io/api/v1/search/?q=domain:{self.word}' response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy) resp = response[0] @@ -32,6 +32,6 @@ class SearchUrlscan: async def get_asns(self) -> List: return self.totalasns - async def process(self, proxy=False): + async def process(self, proxy: bool = False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/virustotal.py b/theHarvester/discovery/virustotal.py index 13bb72dd..33559db1 100644 --- a/theHarvester/discovery/virustotal.py +++ b/theHarvester/discovery/virustotal.py @@ -4,15 +4,15 @@ from theHarvester.lib.core import * class SearchVirustotal: - def __init__(self, word): + def __init__(self, word) -> None: self.key = Core.virustotal_key() if self.key is None: raise MissingKey('virustotal') self.word = word self.proxy = False - self.hostnames = [] + self.hostnames: List = [] - async def do_search(self): + async def do_search(self) -> None: # TODO determine if more endpoints can yield useful info given a domain # based on: https://developers.virustotal.com/reference/domains-relationships # base_url = "https://www.virustotal.com/api/v3/domains/domain/subdomains?limit=40" @@ -81,6 +81,6 @@ class SearchVirustotal: total_subdomains.sort() return total_subdomains - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() diff --git a/theHarvester/discovery/yahoosearch.py b/theHarvester/discovery/yahoosearch.py index 6e94c3b4..92b81609 100644 --- a/theHarvester/discovery/yahoosearch.py +++ b/theHarvester/discovery/yahoosearch.py @@ -4,14 +4,14 @@ from theHarvester.parsers import myparser class SearchYahoo: - def __init__(self, word, limit): + def __init__(self, word, limit) -> None: self.word = word self.total_results = "" self.server = 'search.yahoo.com' self.limit = limit self.proxy = False - async def do_search(self): + async def do_search(self) -> None: base_url = f'https://{self.server}/search?p=%40{self.word}&b=xx&pz=10' headers = { 'Host': self.server, @@ -22,7 +22,7 @@ class SearchYahoo: for response in responses: self.total_results += response - async def process(self): + async def process(self) -> None: await self.do_search() async def get_emails(self): @@ -38,7 +38,7 @@ class SearchYahoo: emails.add(email) return list(emails) - async def get_hostnames(self, proxy=False): + async def get_hostnames(self, proxy: bool=False): self.proxy = proxy rawres = myparser.Parser(self.total_results, self.word) return await rawres.hostnames() diff --git a/theHarvester/discovery/zoomeyesearch.py b/theHarvester/discovery/zoomeyesearch.py index 3d230b5f..d04a9a20 100644 --- a/theHarvester/discovery/zoomeyesearch.py +++ b/theHarvester/discovery/zoomeyesearch.py @@ -1,13 +1,14 @@ from theHarvester.discovery.constants import * from theHarvester.lib.core import * from theHarvester.parsers import myparser +from typing import List import asyncio import re class SearchZoomEye: - def __init__(self, word, limit): + def __init__(self, word, limit) -> None: self.word = word self.limit = limit self.key = Core.zoomeye_key() @@ -19,11 +20,11 @@ class SearchZoomEye: raise MissingKey('zoomeye') self.baseurl = 'https://api.zoomeye.org/host/search' self.proxy = False - self.totalasns = list() - self.totalhosts = list() - self.interestingurls = list() - self.totalips = list() - self.totalemails = list() + self.totalasns: List = list() + self.totalhosts: List = list() + self.interestingurls: List = list() + self.totalips: List = list() + self.totalemails: List = list() # Regex used is directly from: https://github.com/GerbenJavado/LinkFinder/blob/master/linkfinder.py#L29 # Maybe one day it will be a pip package # Regardless LinkFinder is an amazing tool! @@ -56,7 +57,7 @@ class SearchZoomEye: """ self.iurl_regex = re.compile(self.iurl_regex, re.VERBOSE) - async def fetch_subdomains(self): + async def fetch_subdomains(self) -> None: # Based on docs from: https://www.zoomeye.org/doc#search-sub-domain-ip headers = { 'API-KEY': self.key, @@ -91,7 +92,7 @@ class SearchZoomEye: if i % 10 == 0: await asyncio.sleep(get_delay() + 1) - async def do_search(self): + async def do_search(self) -> None: headers = { 'API-KEY': self.key, 'User-Agent': Core.get_user_agent() @@ -211,7 +212,7 @@ class SearchZoomEye: print(f'An exception has occurred: {e}') return hostnames, emails, ips, asns, iurls - async def process(self, proxy=False): + async def process(self, proxy: bool=False) -> None: self.proxy = proxy await self.do_search() # Only need to do it once. diff --git a/theHarvester/lib/api/api.py b/theHarvester/lib/api/api.py index 548022af..0849896c 100644 --- a/theHarvester/lib/api/api.py +++ b/theHarvester/lib/api/api.py @@ -1,5 +1,5 @@ import argparse -from typing import List +from typing import Any, Dict, Union, List import os from fastapi import FastAPI, Header, Query, Request from fastapi.responses import HTMLResponse, UJSONResponse @@ -27,7 +27,7 @@ except RuntimeError: @app.get('/', response_class=HTMLResponse) -async def root(*, user_agent: str = Header(None)): +async def root(*, user_agent: str = Header(None)) -> Union[RedirectResponse, str]: # very basic user agent filtering if user_agent and ('gobuster' in user_agent or 'sqlmap' in user_agent or 'rustbuster' in user_agent): response = RedirectResponse(app.url_path_for('bot')) @@ -59,7 +59,7 @@ async def root(*, user_agent: str = Header(None)): @app.get('/nicebot') -async def bot(): +async def bot() -> Dict[str, str]: # nice bot string = {'bot': 'These are not the droids you are looking for'} return string @@ -77,7 +77,7 @@ async def getsources(request: Request): @app.get('/dnsbrute', response_class=UJSONResponse) @limiter.limit('5/minute') async def dnsbrute(request: Request, user_agent: str = Header(None), - domain: str = Query(..., description='Domain to be brute forced')): + domain: str = Query(..., description='Domain to be brute forced')) -> Union[Dict[str, Any], RedirectResponse]: # Endpoint for user to signal to do DNS brute forcing # Rate limit of 5 requests per minute # basic user agent filtering @@ -112,7 +112,7 @@ async def query(request: Request, dns_server: str = Query(""), user_agent: str = take_over: bool = Query(False), virtual_host: bool = Query(False), source: List[str] = Query(..., description='Data sources to query comma separated with no space'), limit: int = Query(500), start: int = Query(0), - domain: str = Query(..., description='Domain to be harvested')): + domain: str = Query(..., description='Domain to be harvested')) -> Union[Dict[str, Any], RedirectResponse]: # Query function that allows user to query theHarvester rest API # Rate limit of 2 requests per minute diff --git a/theHarvester/lib/api/api_example.py b/theHarvester/lib/api/api_example.py index 1c4761ae..e9799d77 100644 --- a/theHarvester/lib/api/api_example.py +++ b/theHarvester/lib/api/api_example.py @@ -17,7 +17,7 @@ async def fetch(session, url): return await response.text() -async def main(): +async def main() -> None: """ Just a simple example of how to interact with the rest api you can easily use requests instead of aiohttp or whatever you best see fit diff --git a/theHarvester/lib/core.py b/theHarvester/lib/core.py index 2d35e02b..d5efaafb 100644 --- a/theHarvester/lib/core.py +++ b/theHarvester/lib/core.py @@ -1,6 +1,6 @@ # coding=utf-8 from __future__ import annotations -from typing import Union, Any, Tuple, List +from typing import Sized, Union, Any, Tuple, List import yaml import asyncio import aiohttp @@ -238,7 +238,7 @@ class AsyncFetcher: proxy_list = Core.proxy_list() @classmethod - async def post_fetch(cls, url, headers='', data='', params='', json=False, proxy=False): + async def post_fetch(cls, url, headers: Sized='', data: str='', params: str='', json: bool=False, proxy: bool=False): if len(headers) == 0: headers = {'User-Agent': Core.get_user_agent()} timeout = aiohttp.ClientTimeout(total=720) @@ -273,7 +273,7 @@ class AsyncFetcher: return '' @staticmethod - async def fetch(session, url, params='', json=False, proxy="") -> Union[str, dict, list, bool]: + async def fetch(session, url, params: str = '', json: bool = False, proxy: str = "") -> Union[str, dict, list, bool]: # This fetch method solely focuses on get requests try: # Wrap in try except due to 0x89 png/jpg files @@ -306,7 +306,7 @@ class AsyncFetcher: return '' @staticmethod - async def takeover_fetch(session, url, proxy="") -> Union[Tuple[Any, Any], str]: + async def takeover_fetch(session, url: str, proxy: str = "") -> Union[Tuple[Any, Any], str]: # This fetch method solely focuses on get requests try: # Wrap in try except due to 0x89 png/jpg files @@ -326,7 +326,8 @@ class AsyncFetcher: return url, '' @classmethod - async def fetch_all(cls, urls, headers='', params='', json=False, takeover=False, proxy=False) -> tuple: + async def fetch_all(cls, urls, headers: Sized='', params: Sized='', json: bool = False, takeover: bool = False, + proxy: bool = False) -> tuple: # By default, timeout is 5 minutes; 60 seconds should suffice timeout = aiohttp.ClientTimeout(total=60) if len(headers) == 0: diff --git a/theHarvester/lib/hostchecker.py b/theHarvester/lib/hostchecker.py index b8127eb5..82287c4e 100644 --- a/theHarvester/lib/hostchecker.py +++ b/theHarvester/lib/hostchecker.py @@ -8,16 +8,16 @@ Revised to use aiodns & asyncio on 2019-09-23 import aiodns import asyncio import socket -from typing import Tuple, Any +from typing import Tuple, Any, List, Set class Checker: - def __init__(self, hosts: list, nameserver=False): + def __init__(self, hosts: list, nameserver: bool = False) -> None: self.hosts = hosts - self.realhosts: list = [] - self.addresses: set = set() - self.nameserver = [] + self.realhosts: List = [] + self.addresses: Set = set() + self.nameserver: List = [] if nameserver: self.nameserver = nameserver @@ -33,7 +33,8 @@ class Checker: except Exception: return f"{host}", tuple() - async def query_all(self, resolver) -> list: + async def query_all(self, resolver) -> tuple[ + BaseException | Any, BaseException | Any, BaseException | Any, BaseException | Any, BaseException | Any]: results = await asyncio.gather(*[asyncio.create_task(self.query(host, resolver)) for host in self.hosts]) return results diff --git a/theHarvester/lib/stash.py b/theHarvester/lib/stash.py index 2a8ed66d..4a6ebb6f 100644 --- a/theHarvester/lib/stash.py +++ b/theHarvester/lib/stash.py @@ -1,6 +1,8 @@ import aiosqlite import datetime import os +from sqlite3.dbapi2 import Row +from typing import Iterable, Optional, Union, List, Dict db_path = os.path.expanduser('~/.local/share/theHarvester') @@ -10,24 +12,24 @@ if not os.path.isdir(db_path): class StashManager: - def __init__(self): + def __init__(self) -> None: self.db = os.path.join(db_path, 'stash.sqlite') self.results = "" self.totalresults = "" - self.latestscandomain = {} - self.domainscanhistory = [] - self.scanboarddata = {} - self.scanstats = [] - self.latestscanresults = [] - self.previousscanresults = [] + self.latestscandomain: Dict = {} + self.domainscanhistory: List = [] + self.scanboarddata: Dict = {} + self.scanstats: List = [] + self.latestscanresults: List = [] + self.previousscanresults: List = [] - async def do_init(self): + async def do_init(self) -> None: async with aiosqlite.connect(self.db) as db: await db.execute( 'CREATE TABLE IF NOT EXISTS results (domain text, resource text, type text, find_date date, source text)') await db.commit() - async def store(self, domain, resource, res_type, source): + async def store(self, domain, resource, res_type, source) -> None: self.domain = domain self.resource = resource self.type = res_type @@ -41,7 +43,7 @@ class StashManager: except Exception as e: print(e) - async def store_all(self, domain, all, res_type, source): + async def store_all(self, domain, all, res_type, source) -> None: self.domain = domain self.all = all self.type = res_type @@ -109,7 +111,7 @@ class StashManager: except Exception as e: print(e) - async def getlatestscanresults(self, domain, previousday=False): + async def getlatestscanresults(self, domain, previousday: bool=False) -> Optional[Iterable[Union[Row, str]]]: try: async with aiosqlite.connect(self.db, timeout=30) as conn: if previousday: @@ -217,7 +219,7 @@ class StashManager: except Exception as e: print(e) - async def getpluginscanstatistics(self): + async def getpluginscanstatistics(self) -> Optional[Iterable[Row]]: try: async with aiosqlite.connect(self.db, timeout=30) as conn: cursor = await conn.execute(''' diff --git a/theHarvester/parsers/intelxparser.py b/theHarvester/parsers/intelxparser.py index aa56a42a..d61de739 100644 --- a/theHarvester/parsers/intelxparser.py +++ b/theHarvester/parsers/intelxparser.py @@ -1,8 +1,11 @@ +from typing import Set + + class Parser: - def __init__(self): - self.emails = set() - self.hosts = set() + def __init__(self) -> None: + self.emails: Set = set() + self.hosts: Set = set() async def parse_dictionaries(self, results: dict) -> tuple: """ diff --git a/theHarvester/parsers/myparser.py b/theHarvester/parsers/myparser.py index 7a8eb4ac..7b2bb0ae 100644 --- a/theHarvester/parsers/myparser.py +++ b/theHarvester/parsers/myparser.py @@ -1,14 +1,15 @@ import re +from typing import Set, List class Parser: - def __init__(self, results, word): + def __init__(self, results, word) -> None: self.results = results self.word = word - self.temp = [] + self.temp: List = [] - async def genericClean(self): + async def genericClean(self) -> None: self.results = self.results.replace('', '').replace('', '').replace('', '').replace('', '') \ .replace('%3a', '').replace('', '').replace('', '') \ .replace('', '').replace('', '') @@ -16,7 +17,7 @@ class Parser: for search in ('<', '>', ':', '=', ';', '&', '%3A', '%3D', '%3C', '%2f', '/', '\\'): self.results = self.results.replace(search, ' ') - async def urlClean(self): + async def urlClean(self) -> None: self.results = self.results.replace('', '').replace('', '').replace('%2f', '').replace('%3a', '') for search in ('<', '>', ':', '=', ';', '&', '%3A', '%3D', '%3C'): self.results = self.results.replace(search, ' ') @@ -34,7 +35,7 @@ class Parser: return true_emails async def fileurls(self, file): - urls = [] + urls: List = [] reg_urls = re.compile(' Set[str]: found = re.finditer(r'(http|https)://(www\.)?trello.com/([a-zA-Z\d\-_\.]+/?)*', self.results) urls = {match.group().strip() for match in found} return urls diff --git a/theHarvester/parsers/securitytrailsparser.py b/theHarvester/parsers/securitytrailsparser.py index 85ffcde5..bfda60b3 100644 --- a/theHarvester/parsers/securitytrailsparser.py +++ b/theHarvester/parsers/securitytrailsparser.py @@ -1,13 +1,13 @@ -from typing import Union, Tuple, List +from typing import Union, Tuple, List, Set class Parser: - def __init__(self, word, text): + def __init__(self, word, text) -> None: self.word = word self.text = text - self.hostnames = set() - self.ips = set() + self.hostnames: Set = set() + self.ips: Set = set() async def parse_text(self) -> Union[List, Tuple]: sub_domain_flag = 0 diff --git a/theHarvester/screenshot/screenshot.py b/theHarvester/screenshot/screenshot.py index 575f9c07..0e7425ab 100644 --- a/theHarvester/screenshot/screenshot.py +++ b/theHarvester/screenshot/screenshot.py @@ -11,16 +11,17 @@ from datetime import datetime import os import ssl import sys +from typing import Sized, Tuple class ScreenShotter: - def __init__(self, output): + def __init__(self, output) -> None: self.output = output self.slash = "\\" if 'win' in sys.platform else '/' self.slash = "" if (self.output[-1] == "\\" or self.output[-1] == "/") else self.slash - def verify_path(self): + def verify_path(self) -> bool: try: if not os.path.isdir(self.output): answer = input( @@ -36,19 +37,19 @@ class ScreenShotter: return False @staticmethod - async def verify_installation(): + async def verify_installation() -> None: # Helper function that verifies pyppeteer & chromium are installed # If chromium is not installed pyppeteer will prompt user to install it browser = await launch(headless=True, ignoreHTTPSErrors=True, args=["--no-sandbox"]) await browser.close() @staticmethod - def chunk_list(items, chunk_size): + def chunk_list(items: Sized, chunk_size): # Based off of: https://github.com/apache/incubator-sdap-ingester return [items[i:i + chunk_size] for i in range(0, len(items), chunk_size)] @staticmethod - async def visit(url): + async def visit(url: str) -> Tuple[str, str]: try: # print(f'attempting to visit: {url}') timeout = aiohttp.ClientTimeout(total=35) @@ -67,7 +68,7 @@ class ScreenShotter: print(f'An exception has occurred while attempting to visit {url} : {e}') return "", "" - async def take_screenshot(self, url): + async def take_screenshot(self, url: str) -> Tuple[str, ...]: url = f'http://{url}' if not url.startswith('http') else url url = url.replace('www.', '') print(f'Attempting to take a screenshot of: {url}')