diff --git a/.github/workflows/theHarvester.yml b/.github/workflows/theHarvester.yml index e6e34650..5dee3e89 100644 --- a/.github/workflows/theHarvester.yml +++ b/.github/workflows/theHarvester.yml @@ -90,14 +90,6 @@ jobs: run: | python theHarvester.py -d yale.edu -b rapiddns - - name: Run theHarvester module Sublist3r - run: | - python theHarvester.py -d yale.edu -b sublist3r - - - name: Run theHarvester module Threatcrowd - run: | - python theHarvester.py -d yale.edu -b threatcrowd - - name: Run theHarvester module Threatminer run: | python theHarvester.py -d yale.edu -b threatminer diff --git a/Dockerfile b/Dockerfile index e2c7ab91..e982fcaf 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,6 +1,9 @@ -FROM alpine:3.17.0 +FROM alpine:3.17.3 LABEL maintainer="@jay_townsend1 & @NotoriousRebel1 (alpine @viardant)" RUN mkdir /app +RUN mkdir /etc/theHarvester/ +COPY api-keys.yaml /etc/theHarvester/ +COPY proxies.yaml /etc/theHarvester/ WORKDIR /app COPY requirements.txt requirements.txt COPY requirements requirements diff --git a/docker-compose.yml b/docker-compose.yml index 4850c2ba..46b9fdcd 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -4,6 +4,8 @@ services: container_name: theHarvester volumes: - ./api-keys.yaml:/app/api-keys.yaml + - ./api-keys.yaml:/etc/theHarvester/api-keys.yaml + - ./proxies.yaml:/etc/theHarvester/proxies.yaml build: . ports: - "8080:80" diff --git a/requirements/dev.txt b/requirements/dev.txt index f80e09ff..c4ab7d62 100644 --- a/requirements/dev.txt +++ b/requirements/dev.txt @@ -2,6 +2,7 @@ flake8==6.0.0 mypy==1.2.0 mypy-extensions==1.0.0 +pydantic==1.10.7 pyre-check==0.9.18 pyflakes==3.0.1 pytest==7.2.2 @@ -11,4 +12,5 @@ types-chardet==5.0.4.3 types-ujson==5.7.0.1 types-PyYAML==6.0.12.9 types-requests==2.28.11.17 +types-python-dateutil==2.8.19.12 wheel==0.40.0 \ No newline at end of file diff --git a/setup.cfg b/setup.cfg index 777ae408..fc9e7a60 100644 --- a/setup.cfg +++ b/setup.cfg @@ -1,2 +1,2 @@ [flake8] -ignore = E501, F405, F403, E402, F401 \ No newline at end of file +ignore = E501, F405, F403, E402, F401, F402 \ No newline at end of file diff --git a/theHarvester/__main__.py b/theHarvester/__main__.py index ae31d612..9390693d 100644 --- a/theHarvester/__main__.py +++ b/theHarvester/__main__.py @@ -18,7 +18,8 @@ import secrets async def start(rest_args: Optional[argparse.Namespace] = None): """Main program function""" - parser = argparse.ArgumentParser(description='theHarvester is used to gather open source intelligence (OSINT) on a company or domain.') + parser = argparse.ArgumentParser( + description='theHarvester is used to gather open source intelligence (OSINT) on a company or domain.') parser.add_argument('-d', '--domain', help='Company name or domain to search.', required=True) parser.add_argument('-l', '--limit', help='Limit the number of search results, default=500.', default=500, type=int) parser.add_argument('-S', '--start', help='Start with result number X, default=0.', default=0, type=int) @@ -29,15 +30,15 @@ async def start(rest_args: Optional[argparse.Namespace] = None): parser.add_argument('-e', '--dns-server', help='DNS server to use for lookup.') parser.add_argument('-t', '--take-over', help='Check for takeovers.', default=False, action='store_true') # TODO add dns resolver flag - parser.add_argument('-r', '--dns-resolve', help='Perform DNS resolution on subdomains with given resolver list or passed in resolvers, default False.', default="", + parser.add_argument('-r', '--dns-resolve', help='Perform DNS resolution on subdomains with a resolver list or passed in resolvers, default False.', default="", type=str, nargs='?') parser.add_argument('-n', '--dns-lookup', help='Enable DNS server lookup, default False.', default=False, action='store_true') parser.add_argument('-c', '--dns-brute', help='Perform a DNS brute force on the domain.', default=False, action='store_true') parser.add_argument('-f', '--filename', help='Save the results to an XML and JSON file.', default='', type=str) - parser.add_argument('-b', '--source', help='''anubis, baidu, bevigil, binaryedge, bing, bingapi, bufferoverun, brave, - censys, certspotter, criminalip, crtsh, dnsdumpster, duckduckgo, fullhunt, github-code, - hackertarget, hunter, hunterhow, intelx, otx, pentesttools, projectdiscovery, qwant, - rapiddns, rocketreach, securityTrails, subdomainfinderc99, threatminer, urlscan, + parser.add_argument('-b', '--source', help='''anubis, baidu, bevigil, binaryedge, bing, bingapi, bufferoverun, brave, + censys, certspotter, criminalip, crtsh, dnsdumpster, duckduckgo, fullhunt, github-code, + hackertarget, hunter, hunterhow, intelx, otx, pentesttools, projectdiscovery, qwant, + rapiddns, rocketreach, securityTrails, subdomainfinderc99, threatminer, urlscan, virustotal, yahoo, zoomeye''') # determines if filename is coming from rest api or user @@ -73,7 +74,7 @@ async def start(rest_args: Optional[argparse.Namespace] = None): all_hosts: List = [] all_ip: List = [] dnslookup = args.dns_lookup - dnsserver = args.dns_server # TODO arg is not used anywhere replace with resolvers wordlist arg dnsresolve + dnsserver = args.dns_server # TODO arg is not used anywhere replace with resolvers wordlist arg dnsresolve dnsresolve = args.dns_resolve final_dns_resolver_list = [] if dnsresolve is not None and len(dnsresolve) > 0: @@ -109,9 +110,8 @@ async def start(rest_args: Optional[argparse.Namespace] = None): print(f'Dumping resolvers passed in: {e}') sys.exit(0) - # if for some reason there are duplicates + # if for some reason, there are duplicates final_dns_resolver_list = list(set(final_dns_resolver_list)) - # print(f'My final list: {final_dns_resolver_list}') engines: List = [] # If the user specifies @@ -146,7 +146,7 @@ async def start(rest_args: Optional[argparse.Namespace] = None): store_interestingurls: bool = False, store_asns: bool = False) -> None: """ Persist details into the database. - The details to be stored is controlled by the parameters passed to the method. + The details to be stored are controlled by the parameters passed to the method. :param search_engine: search engine to fetch details from :param source: source against which the details (corresponding to the search engine) need to be persisted @@ -163,12 +163,14 @@ async def start(rest_args: Optional[argparse.Namespace] = None): await search_engine.process(use_proxy) if process_param is None else await \ search_engine.process(process_param, use_proxy) db_stash = stash.StashManager() + if source: print(f'\033[94m[*] Searching {source[0].upper() + source[1:]}. ') + if store_host: host_names = [host for host in filter(await search_engine.get_hostnames()) if f'.{word}' in host] if source != 'hackertarget' and source != 'pentesttools' and source != 'rapiddns': - # If source is inside this conditional it means the hosts returned must be resolved to obtain ip + # If a source is inside this conditional, it means the hosts returned must be resolved to obtain ip # This should only be checked if --dns-resolve has a wordlist if dnsresolve is None or len(final_dns_resolver_list) > 0: # indicates that -r was passed in @@ -182,14 +184,17 @@ async def start(rest_args: Optional[argparse.Namespace] = None): full.extend(host_names) all_hosts.extend(host_names) await db_stash.store_all(word, all_hosts, 'host', source) + if store_emails: email_list = filter(await search_engine.get_emails()) all_emails.extend(email_list) await db_stash.store_all(word, email_list, 'email', source) + if store_ip: ips_list = await search_engine.get_ips() all_ip.extend(ips_list) await db_stash.store_all(word, all_ip, 'ip', source) + if store_results: email_list, host_names, urls = await search_engine.get_results() all_emails.extend(email_list) @@ -198,19 +203,23 @@ async def start(rest_args: Optional[argparse.Namespace] = None): all_hosts.extend(host_names) await db.store_all(word, all_hosts, 'host', source) await db.store_all(word, all_emails, 'email', source) + if store_people: people_list = await search_engine.get_people() await db_stash.store_all(word, people_list, 'people', source) + if store_links: links = await search_engine.get_links() linkedin_links_tracker.extend(links) if len(links) > 0: await db.store_all(word, links, 'linkedinlinks', engineitem) + if store_interestingurls: iurls = await search_engine.get_interestingurls() interesting_urls.extend(iurls) if len(iurls) > 0: await db.store_all(word, iurls, 'interestingurls', engineitem) + if store_asns: fasns = await search_engine.get_asns() total_asns.extend(fasns) @@ -664,7 +673,8 @@ async def start(rest_args: Optional[argparse.Namespace] = None): all_hosts = list(sorted(list(set(all_hosts)))) db = stash.StashManager() all_hosts = [host.replace('www.', '') for host in all_hosts] - full = [host if ':' in host and host.endswith(word) else host.split(':')[0].endswith(word) and host for host in full] + full = [host if ':' in host and host.endswith(word) else host.split(':')[0].endswith(word) and host for host in + full] full = list({host.replace('www.', '') for host in full if host}) full.sort(key=lambda el: el.split(':')[0]) for host in full: @@ -719,8 +729,8 @@ async def start(rest_args: Optional[argparse.Namespace] = None): target=word, local_results=dnsrev, overall_results=full), - nameservers= final_dns_resolver_list if len(final_dns_resolver_list) > 0 else None)) - #nameservers=list(map(str, dnsserver.split(','))) if dnsserver else None)) + nameservers=final_dns_resolver_list if len(final_dns_resolver_list) > 0 else None)) + # nameservers=list(map(str, dnsserver.split(','))) if dnsserver else None)) # run all the reversing tasks concurrently await asyncio.gather(*__reverse_dns_tasks.values()) @@ -741,7 +751,7 @@ async def start(rest_args: Optional[argparse.Namespace] = None): await basic_search.process_vhost() results = await basic_search.get_allhostnames() for result in results: - result = re.sub(r'[[]*', '', result) + result = re.sub(r'[[]*', '', result) result = re.sub('<', '', result) result = re.sub('>', '', result) print((data + '\t' + result)) @@ -765,7 +775,7 @@ async def start(rest_args: Optional[argparse.Namespace] = None): print(f'\nScreenshots can be found in: {screen_shotter.output}{screen_shotter.slash}') start_time = time.perf_counter() print('Filtering domains for ones we can reach') - unique_resolved_domains = {url.split(':')[0]for url in full if ':' in url and 'www.' not in url} + unique_resolved_domains = {url.split(':')[0] for url in full if ':' in url and 'www.' not in url} if len(unique_resolved_domains) > 0: # First filter out ones that didn't resolve print('Attempting to visit unique resolved domains, this is ACTIVE RECON') @@ -775,7 +785,7 @@ async def start(rest_args: Optional[argparse.Namespace] = None): unique_resolved_domains = list(sorted({tup[0] for tup in results if len(tup[1]) > 0})) async with Pool(3) as pool: print(f'Length of unique resolved domains: {len(unique_resolved_domains)} chunking now!\n') - # If you have the resources you could make the function faster by increasing the chunk number + # If you have the resources, you could make the function faster by increasing the chunk number chunk_number = 14 for chunk in screen_shotter.chunk_list(unique_resolved_domains, chunk_number): try: @@ -909,4 +919,4 @@ async def entry_point() -> None: print('\n\n[!] ctrl+c detected from user, quitting.\n\n ') except Exception as error_entry_point: print(error_entry_point) - sys.exit(1) \ No newline at end of file + sys.exit(1) diff --git a/theHarvester/discovery/bravesearch.py b/theHarvester/discovery/bravesearch.py index c87bb86f..80a307b7 100644 --- a/theHarvester/discovery/bravesearch.py +++ b/theHarvester/discovery/bravesearch.py @@ -16,9 +16,9 @@ class SearchBrave: async def do_search(self): try: headers = {'User-Agent': Core.get_user_agent()} - for i in range(0, 50): + for offset in range(0, 50): # To reduce total number of requests made just search for "self.word" instead of self.word - current_url = f'{self.server}"{self.word}"&offset={i}&source=web&show_local=0' + current_url = f'{self.server}"{self.word}"&offset={offset}&source=web&show_local=0' resp = await AsyncFetcher.fetch_all([current_url], headers=headers, proxy=self.proxy) self.results = resp[0] self.totalresults += self.results @@ -26,8 +26,8 @@ class SearchBrave: if 'Not many great matches came back for your search' in resp[0] \ or 'Your request has been flagged as being suspicious and Brave Search' in resp[0] \ or 'Prove' in resp[0] and 'robot' in resp[0] or 'Robot' in resp[0]: - # print('Breaking!') - break + # print('Breaking!') + break await asyncio.sleep(get_delay() + 5) except Exception as e: print(f'An exception has occurred in bravesearch: {e}') diff --git a/theHarvester/discovery/criminalip.py b/theHarvester/discovery/criminalip.py index 4a252220..ea13f89f 100644 --- a/theHarvester/discovery/criminalip.py +++ b/theHarvester/discovery/criminalip.py @@ -20,7 +20,7 @@ class SearchCriminalIP: # https://www.criminalip.io/developer/api/post-domain-scan # https://www.criminalip.io/developer/api/get-domain-status-id # https://www.criminalip.io/developer/api/get-domain-report-id - url = f'https://api.criminalip.io/v1/domain/scan' + url = 'https://api.criminalip.io/v1/domain/scan' data = f'{{"query": "{self.word}"}}' # print(f'Current key: {self.key}') user_agent = Core.get_user_agent() diff --git a/theHarvester/discovery/duckduckgosearch.py b/theHarvester/discovery/duckduckgosearch.py index 17c54315..8df3e9c2 100644 --- a/theHarvester/discovery/duckduckgosearch.py +++ b/theHarvester/discovery/duckduckgosearch.py @@ -1,8 +1,8 @@ +from pydantic.types import Json from theHarvester.discovery.constants import * from theHarvester.lib.core import * from theHarvester.parsers import myparser import json -from typing import Union class SearchDuckDuckGo: @@ -31,7 +31,7 @@ class SearchDuckDuckGo: all_resps = await AsyncFetcher.fetch_all(urls) self.totalresults += ''.join(all_resps) - async def crawl(self, text: Union[bytes, str]): + async def crawl(self, text: Json): """ Function parses json and returns URLs. :param text: formatted json @@ -42,20 +42,24 @@ class SearchDuckDuckGo: load = json.loads(text) for keys in load.keys(): # Iterate through keys of dict. val = load.get(keys) + if isinstance(val, int) or isinstance(val, dict) or val is None: continue + if isinstance(val, list): if len(val) == 0: # Make sure not indexing an empty list. continue val = val[0] # The First value should be dict. + if isinstance(val, dict): # Validation check. for key in val.keys(): value = val.get(key) if isinstance(value, str) and value != '' and 'https://' in value or 'http://' in value: urls.add(value) + if isinstance(val, str) and val != '' and 'https://' in val or 'http://' in val: urls.add(val) - tmp = set() + tmp = set for url in urls: if '<' in url and 'href=' in url: # Format is equal_index = url.index('=') diff --git a/theHarvester/discovery/githubcode.py b/theHarvester/discovery/githubcode.py index 9d1511c6..6479ba96 100644 --- a/theHarvester/discovery/githubcode.py +++ b/theHarvester/discovery/githubcode.py @@ -14,8 +14,8 @@ class RetryResult(NamedTuple): class SuccessResult(NamedTuple): fragments: List[str] - next_page: Optional[int] - last_page: Optional[int] + next_page: Optional[Any] + last_page: Optional[Any] class ErrorResult(NamedTuple): @@ -33,7 +33,7 @@ class SearchGithubCode: self.counter: int = 0 self.page: int = 1 self.key = Core.github_key() - # If you don't have a personal access token, github narrows your search capabilities significantly + # If you don't have a personal access token, GitHub narrows your search capabilities significantly # rate limits you more severely # https://developer.github.com/v3/search/#rate-limit if self.key is None: diff --git a/theHarvester/discovery/subdomainfinderc99.py b/theHarvester/discovery/subdomainfinderc99.py index 9d85f215..d13139f9 100644 --- a/theHarvester/discovery/subdomainfinderc99.py +++ b/theHarvester/discovery/subdomainfinderc99.py @@ -54,7 +54,7 @@ class SearchSubdomainfinderc99: for c in html.find_all('input'): try: csrf_params[c.get('name')] = c.get('value') - except: + except Exception: continue return csrf_params diff --git a/theHarvester/discovery/sublist3r.py b/theHarvester/discovery/sublist3r.py deleted file mode 100644 index ee0611ae..00000000 --- a/theHarvester/discovery/sublist3r.py +++ /dev/null @@ -1,22 +0,0 @@ -from typing import Type -from theHarvester.lib.core import * - - -class SearchSublist3r: - - def __init__(self, word): - self.word = word - self.totalhosts = list - self.proxy = False - - async def do_search(self): - url = f'https://api.sublist3r.com/search.php?domain={self.word}' - response = await AsyncFetcher.fetch_all([url], json=True, proxy=self.proxy) - self.totalhosts: list = response[0] - - async def get_hostnames(self) -> Type[list]: - return self.totalhosts - - async def process(self, proxy=False): - self.proxy = proxy - await self.do_search() diff --git a/theHarvester/discovery/takeover.py b/theHarvester/discovery/takeover.py index d06404be..9bc0b445 100644 --- a/theHarvester/discovery/takeover.py +++ b/theHarvester/discovery/takeover.py @@ -49,9 +49,9 @@ class TakeOver: async def do_take(self) -> None: try: if len(self.hosts) > 0: - tup_resps: list = await AsyncFetcher.fetch_all(self.hosts, takeover=True, proxy=self.proxy) + tup_resps: tuple = await AsyncFetcher.fetch_all(self.hosts, takeover=True, proxy=self.proxy) # Returns a list of tuples in this format: (url, response) - tup_resps = [tup for tup in tup_resps if tup[1] != ''] + tup_resps = tuple(tup for tup in tup_resps if tup[1] != '') # Filter out responses whose responses are empty strings (indicates errored) for url, resp in tup_resps: await self.check(url, resp) diff --git a/theHarvester/discovery/threatcrowd.py b/theHarvester/discovery/threatcrowd.py deleted file mode 100644 index 78cbbfc3..00000000 --- a/theHarvester/discovery/threatcrowd.py +++ /dev/null @@ -1,33 +0,0 @@ -from typing import List -from theHarvester.lib.core import * - - -class SearchThreatcrowd: - - def __init__(self, word): - self.word = word.replace(' ', '%20') - self.hostnames = list() - self.ips = list() - self.proxy = False - - async def do_search(self): - base_url = f'https://www.threatcrowd.org/searchApi/v2/domain/report/?domain={self.word}' - headers = {'User-Agent': Core.get_user_agent()} - try: - responses = await AsyncFetcher.fetch_all([base_url], headers=headers, proxy=self.proxy, json=True) - resp = responses[0] - self.ips = {ip['ip_address'] for ip in resp['resolutions'] if len(ip['ip_address']) > 4} - self.hostnames = set(list(resp['subdomains'])) - except Exception as e: - print(e) - - async def get_ips(self) -> List: - return self.ips - - async def get_hostnames(self) -> List: - return self.hostnames - - async def process(self, proxy=False): - self.proxy = proxy - await self.do_search() - await self.get_hostnames() diff --git a/theHarvester/discovery/virustotal.py b/theHarvester/discovery/virustotal.py index 00342653..edc4d82d 100644 --- a/theHarvester/discovery/virustotal.py +++ b/theHarvester/discovery/virustotal.py @@ -84,4 +84,4 @@ class SearchVirustotal: async def process(self, proxy: bool = False) -> None: self.proxy = proxy - await self.do_search() \ No newline at end of file + await self.do_search() diff --git a/theHarvester/lib/core.py b/theHarvester/lib/core.py index 121407ff..0b4ffc21 100644 --- a/theHarvester/lib/core.py +++ b/theHarvester/lib/core.py @@ -243,11 +243,11 @@ class AsyncFetcher: if len(headers) == 0: headers = {'User-Agent': Core.get_user_agent()} timeout = aiohttp.ClientTimeout(total=720) - # By default, timeout is 5 minutes, changed to 12 minutes + # By default, timeout is 5 minutes, changed to 12-minutes # results are well worth the wait try: if proxy: - proxy = str(random.choice(cls().proxy_list)) + proxy = random.choice(cls().proxy_list) if params != "": async with aiohttp.ClientSession(headers=headers, timeout=timeout) as session: async with session.get(url, params=params, proxy=proxy) as response: @@ -279,7 +279,7 @@ class AsyncFetcher: @classmethod async def fetch(cls, session, url, params: str = '', json: bool = False, proxy: str = "") -> Union[ - str, dict, list, bool]: + str, dict, list, bool]: # This fetch method solely focuses on get requests try: # Wrap in try except due to 0x89 png/jpg files diff --git a/theHarvester/lib/hostchecker.py b/theHarvester/lib/hostchecker.py index 60718a7e..61b1e807 100644 --- a/theHarvester/lib/hostchecker.py +++ b/theHarvester/lib/hostchecker.py @@ -34,7 +34,7 @@ class Checker: return f"{host}", tuple() async def query_all(self, resolver) -> tuple[ - BaseException | Any, BaseException | Any, BaseException | Any, BaseException | Any, BaseException | Any]: + BaseException | Any, BaseException | Any, BaseException | Any, BaseException | Any, BaseException | Any]: results = await asyncio.gather(*[asyncio.create_task(self.query(host, resolver)) for host in self.hosts]) return results diff --git a/theHarvester/lib/stash.py b/theHarvester/lib/stash.py index b9d13cad..4492e08b 100644 --- a/theHarvester/lib/stash.py +++ b/theHarvester/lib/stash.py @@ -121,7 +121,7 @@ class StashManager: FROM results WHERE find_date=date('now', '-1 day') and domain=?''', (domain,)) previousscandate = await cursor.fetchone() - if not previousscandate: # When theHarvester runs first time/day this query will return. + if not previousscandate: # When theHarvester runs first time/day, this query will return. self.previousscanresults = ["No results", "No results", "No results", "No results", "No results"] else: @@ -153,6 +153,7 @@ class StashManager: print(f'Error in getting the latest scan results from the database: {e}') except Exception as e: print(f'Error connecting to theHarvester database: {e}') + return self.latestscanresults async def getscanboarddata(self): try: @@ -229,9 +230,9 @@ class StashManager: ''') results = await cursor.fetchall() self.scanstats = results - return self.scanstats except Exception as e: print(e) + return self.scanstats async def latestscanchartdata(self, domain): try: diff --git a/theHarvester/parsers/myparser.py b/theHarvester/parsers/myparser.py index 53631c1d..5de16a24 100644 --- a/theHarvester/parsers/myparser.py +++ b/theHarvester/parsers/myparser.py @@ -56,7 +56,7 @@ class Parser: # TODO determine if necessary below or if only pass through is fine reg_hosts = re.compile(r'[a-zA-Z0-9.-]*\.' + self.word.replace('www.', '')) # reg_hosts = re.compile(r'www\.[a-zA-Z0-9.-]*\.' + 'www.' + self.word) - #reg_hosts = re.compile(r'www\.[a-zA-Z0-9.-]*\.(?:' + 'www.' + self.word + ')?') + # reg_hosts = re.compile(r'www\.[a-zA-Z0-9.-]*\.(?:' + 'www.' + self.word + ')?') second_hostnames = reg_hosts.findall(self.results) hostnames.extend(second_hostnames) return list(set(hostnames))