From d56ee5a67288174305c1cfffc4079e61d297e995 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Thu, 12 Sep 2019 17:45:50 -0400 Subject: [PATCH 01/12] Fixed two mypy errors. --- theHarvester/discovery/constants.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/theHarvester/discovery/constants.py b/theHarvester/discovery/constants.py index cfa81346..c0337a50 100644 --- a/theHarvester/discovery/constants.py +++ b/theHarvester/discovery/constants.py @@ -1,3 +1,4 @@ +from typing import Union import random googleUA = 'Mozilla/5.0 (Windows NT 6.2) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/28.0.1464.0 Safari/537.36' @@ -47,7 +48,7 @@ def filter(lst): return new_lst -def getDelay() -> int: +def getDelay() -> float: return random.randint(1, 3) - .5 @@ -61,7 +62,7 @@ def search(text: str) -> bool: return False -def google_workaround(visit_url: str) -> str or bool: +def google_workaround(visit_url: str) -> Union[bool, list]: """ Function that makes a request on our behalf, if Google starts to block us :param visit_url: Url to scrape From 24d2d2c9036e672310fa425d7931d22fcb2b5fae Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Sun, 22 Sep 2019 02:21:08 -0400 Subject: [PATCH 02/12] Reworked core logic for trello, changed how main.py worked with trello, and updated two functions in myparser. --- theHarvester/__main__.py | 30 +++++++------ theHarvester/discovery/constants.py | 2 +- theHarvester/discovery/trello.py | 65 +++++++++++++++++++---------- theHarvester/parsers/myparser.py | 11 ++--- 4 files changed, 64 insertions(+), 44 deletions(-) diff --git a/theHarvester/__main__.py b/theHarvester/__main__.py index 6ac1188e..43bab578 100644 --- a/theHarvester/__main__.py +++ b/theHarvester/__main__.py @@ -61,7 +61,7 @@ def start(): shodan = args.shodan start = args.start # type: int takeover_check = False - trello_info = ([], False) + trello_urls = [] vhost = [] virtual = args.virtual_host word = args.domain # type: str @@ -345,13 +345,12 @@ def start(): print('\033[94m[*] Searching Trello. \033[0m') from theHarvester.discovery import trello # Import locally or won't work. - trello_search = trello.SearchTrello(word, limit) + trello_search = trello.SearchTrello(word) trello_search.process() - emails = filter(trello_search.get_emails()) + emails, hosts, urls = trello_search.get_results() all_emails.extend(emails) - info = trello_search.get_urls() - hosts = filter(info[0]) - trello_info = (info[1], True) + hosts = filter(hosts) + trello_urls = filter(urls) all_hosts.extend(hosts) db = stash.stash_manager() db.store_all(word, hosts, 'host', 'trello') @@ -448,16 +447,15 @@ def start(): db = stash.stash_manager() db.store_all(word, host_ip, 'ip', 'DNS-resolver') - if trello_info[1] is True: - trello_urls = trello_info[0] - if trello_urls is []: - print('\n[*] No URLs found.') - else: - total = len(trello_urls) - print('\n[*] URLs found: ' + str(total)) - print('--------------------') - for url in sorted(list(set(trello_urls))): - print(url) + length_urls = len(trello_urls) + if length_urls == 0: + print('\n[*] No Trello URLs found.') + else: + total = length_urls + print('\n[*] Trello URLs found: ' + str(total)) + print('--------------------') + for url in sorted(trello_urls): + print(url) # DNS brute force # dnsres = [] diff --git a/theHarvester/discovery/constants.py b/theHarvester/discovery/constants.py index 5244f21b..0cd242f8 100644 --- a/theHarvester/discovery/constants.py +++ b/theHarvester/discovery/constants.py @@ -62,7 +62,7 @@ def search(text: str) -> bool: return False -def google_workaround(visit_url: str) -> Union[bool, list]: +def google_workaround(visit_url: str) -> Union[bool, str]: """ Function that makes a request on our behalf, if Google starts to block us :param visit_url: Url to scrape diff --git a/theHarvester/discovery/trello.py b/theHarvester/discovery/trello.py index bda0d0c5..2718139e 100644 --- a/theHarvester/discovery/trello.py +++ b/theHarvester/discovery/trello.py @@ -1,30 +1,44 @@ from theHarvester.discovery.constants import * from theHarvester.parsers import myparser import grequests - +import requests +import random +import time class SearchTrello: - def __init__(self, word, limit): + def __init__(self, word): self.word = word.replace(' ', '%20') self.results = "" self.totalresults = "" self.server = 'www.google.com' self.quantity = '100' - self.limit = limit + self.limit = 300 + self.trello_urls = [] + self.hostnames = [] self.counter = 0 def do_search(self): - base_url = f'https://{self.server}/search?num=100&start=xx&hl=en&q=site%3Atrello.com%20{self.word}' + base_url = f'https://{self.server}/search?num=300&start=xx&hl=en&q=site%3Atrello.com%20{self.word}' + urls = [base_url.replace("xx", str(num)) for num in range(0, self.limit, 20) if num <= self.limit] + # limit is 20 as that is the most results google will show per num headers = {'User-Agent': googleUA} - try: - urls = [base_url.replace("xx", str(num)) for num in range(0, self.limit, 10) if num <= self.limit] - request = (grequests.get(url, headers=headers) for url in urls) - response = grequests.imap(request, size=5) - for entry in response: - self.totalresults += entry.content.decode('UTF-8') - except Exception as e: - print(e) + for url in urls: + try: + resp = requests.get(url, headers=headers) + self.results = resp.text + if search(self.results): + try: + self.results = google_workaround(base_url) + if isinstance(self.results, bool): + print('Google is blocking your ip and the workaround, returning') + return + except Exception as e: + print(e) + self.totalresults += self.results + time.sleep(getDelay()-.5) + except Exception as e: + pass def get_emails(self): rawres = myparser.Parser(self.totalresults, self.word) @@ -33,17 +47,18 @@ class SearchTrello: def get_urls(self): try: rawres = myparser.Parser(self.totalresults, 'trello.com') - trello_urls = rawres.urls() - visited = set() - for url in trello_urls: - # Iterate through Trello URLs gathered and visit them, append text to totalresults. - if url not in visited: # Make sure visiting unique URLs. - visited.add(url) - request = grequests.get(url=url, headers={'User-Agent': googleUA}) - response = grequests.map([request]) - self.totalresults = response[0].content.decode('UTF-8') + self.trello_urls = set(rawres.urls()) + self.totalresults = '' + # reset what totalresults as before it was just google results now it is trello results + headers = {'User-Agent': random.choice(['curl/7.37.0', 'Wget/1.19.4'])} + # do not change the headers + req = (grequests.get(url, headers=headers, timeout=4) for url in self.trello_urls) + responses = grequests.imap(req, size=8) + for response in responses: + self.totalresults += response.content.decode('UTF-8') + rawres = myparser.Parser(self.totalresults, self.word) - return rawres.hostnames(), trello_urls + self.hostnames = rawres.hostnames() except Exception as e: print(f'Error occurred: {e}') @@ -51,3 +66,9 @@ class SearchTrello: self.do_search() self.get_urls() print(f'\tSearching {self.counter} results.') + + def get_results(self) -> tuple: + return self.get_emails(), self.hostnames, self.trello_urls + + + diff --git a/theHarvester/parsers/myparser.py b/theHarvester/parsers/myparser.py index 83a843c1..d1bfff1e 100644 --- a/theHarvester/parsers/myparser.py +++ b/theHarvester/parsers/myparser.py @@ -25,7 +25,7 @@ class Parser: self.genericClean() # Local part is required, charset is flexible. # https://tools.ietf.org/html/rfc6531 (removed * and () as they provide FP mostly) - reg_emails = re.compile(r'[a-zA-Z0-9.\-_+#~!$&\',;=:]+' + '@' + '[a-zA-Z0-9.-]*' + self.word) + reg_emails = re.compile(r'[a-zA-Z0-9.\-_+#~!$&\',;=:]+' + '@' + '[a-zA-Z0-9.-]*' + self.word.replace('www.', '')) self.temp = reg_emails.findall(self.results) emails = self.unique() return emails @@ -47,6 +47,9 @@ class Parser: reg_hosts = re.compile(r'[a-zA-Z0-9.-]*\.' + self.word) self.temp = reg_hosts.findall(self.results) hostnames = self.unique() + reg_hosts = re.compile(r'[a-zA-Z0-9.-]*\.' + self.word.replace('www.', '')) + self.temp = reg_hosts.findall(self.results) + hostnames.extend(self.unique()) return hostnames def people_googleplus(self): @@ -138,10 +141,8 @@ class Parser: return sets def urls(self): - found = re.finditer(r'https://(www\.)?trello.com/([a-zA-Z0-9\-_\.]+/?)*', self.results) - for x in found: - self.temp.append(x.group()) - urls = self.unique() + found = re.finditer(r'(http|https)://(www\.)?trello.com/([a-zA-Z0-9\-_\.]+/?)*', self.results) + urls = {match.group().strip() for match in found} return urls def unique(self): From 7fc1201318a75505381de9e65f4b38fd1fbf3e58 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Sun, 22 Sep 2019 13:38:56 -0400 Subject: [PATCH 03/12] Updated trello search to fix pep8 errors. --- theHarvester/discovery/trello.py | 6 ++---- theHarvester/parsers/myparser.py | 4 ++-- 2 files changed, 4 insertions(+), 6 deletions(-) diff --git a/theHarvester/discovery/trello.py b/theHarvester/discovery/trello.py index 2718139e..2b2ff7b1 100644 --- a/theHarvester/discovery/trello.py +++ b/theHarvester/discovery/trello.py @@ -5,6 +5,7 @@ import requests import random import time + class SearchTrello: def __init__(self, word): @@ -36,7 +37,7 @@ class SearchTrello: except Exception as e: print(e) self.totalresults += self.results - time.sleep(getDelay()-.5) + time.sleep(getDelay() - .5) except Exception as e: pass @@ -69,6 +70,3 @@ class SearchTrello: def get_results(self) -> tuple: return self.get_emails(), self.hostnames, self.trello_urls - - - diff --git a/theHarvester/parsers/myparser.py b/theHarvester/parsers/myparser.py index d1bfff1e..ddc03c94 100644 --- a/theHarvester/parsers/myparser.py +++ b/theHarvester/parsers/myparser.py @@ -50,7 +50,7 @@ class Parser: reg_hosts = re.compile(r'[a-zA-Z0-9.-]*\.' + self.word.replace('www.', '')) self.temp = reg_hosts.findall(self.results) hostnames.extend(self.unique()) - return hostnames + return list(set(hostnames)) def people_googleplus(self): self.results = re.sub('', '', self.results) @@ -145,7 +145,7 @@ class Parser: urls = {match.group().strip() for match in found} return urls - def unique(self): + def unique(self) -> list: self.new = [] for x in self.temp: if x not in self.new: From 8e3bc4b3b99591ca184ff39953d80f58ee94ad00 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Sun, 22 Sep 2019 13:59:13 -0400 Subject: [PATCH 04/12] Updated trello to conform to pep8 standards. --- theHarvester/discovery/trello.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/theHarvester/discovery/trello.py b/theHarvester/discovery/trello.py index 2b2ff7b1..c3d1c5ad 100644 --- a/theHarvester/discovery/trello.py +++ b/theHarvester/discovery/trello.py @@ -39,7 +39,7 @@ class SearchTrello: self.totalresults += self.results time.sleep(getDelay() - .5) except Exception as e: - pass + print(f'An exception has occurred in trello: {e}') def get_emails(self): rawres = myparser.Parser(self.totalresults, self.word) From 8348b5cd022c2d357fd44033138de6e7bdee39c5 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Sun, 22 Sep 2019 14:12:24 -0400 Subject: [PATCH 05/12] Updated travis config to test matrices. --- .travis.yml | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/.travis.yml b/.travis.yml index d56623c5..d1f7b8e1 100644 --- a/.travis.yml +++ b/.travis.yml @@ -1,9 +1,14 @@ dist: bionic language: python -python: -- '3.6' -- '3.7' -- '3.8-dev' + +matrix: + include: + - python: '3.6' + env: TEST_SUITE=suite_3_6 + - python: '3.7' + env: TEST_SUITE=suite_3_7 + - python: '3.8-dev' + env: TEST_SUITE=suite_3_8_dev before_install: - pip install -r requirements.txt install: From 3e816459bc1166652c8fc375886147ceb0860bc6 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Sun, 22 Sep 2019 14:27:49 -0400 Subject: [PATCH 06/12] Made change to main.py to iterate through search engines in order. --- theHarvester/__main__.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/theHarvester/__main__.py b/theHarvester/__main__.py index 43bab578..f0dc79b7 100644 --- a/theHarvester/__main__.py +++ b/theHarvester/__main__.py @@ -67,8 +67,8 @@ def start(): word = args.domain # type: str if args.source is not None: - engines = set(map(str.strip, args.source.split(','))) - + engines = sorted(set(map(str.strip, args.source.split(',')))) + # Iterate through search engines in order if set(engines).issubset(Core.get_supportedengines()): print(f'\033[94m[*] Target: {word} \n \033[0m') From fafa451c34eb1ff78f728f791c8ded94139e2c08 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Sun, 22 Sep 2019 14:30:14 -0400 Subject: [PATCH 07/12] Updated travis to cache pip. --- .travis.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.travis.yml b/.travis.yml index d1f7b8e1..8ebe51b8 100644 --- a/.travis.yml +++ b/.travis.yml @@ -1,6 +1,6 @@ dist: bionic language: python - +cache: pip matrix: include: - python: '3.6' From 15e61523a41aacfe9ab3be6ba315d01825a6e416 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Sun, 22 Sep 2019 14:46:09 -0400 Subject: [PATCH 08/12] Set limit to 200. --- .travis.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.travis.yml b/.travis.yml index 8ebe51b8..fd7d818a 100644 --- a/.travis.yml +++ b/.travis.yml @@ -14,7 +14,7 @@ before_install: install: - python setup.py test script: -- python theHarvester.py -d metasploit.com -b baidu,bing,censys,crtsh,dnsdumpster,dogpile,duckduckgo,exalead,linkedin,netcraft,threatcrowd,trello,twitter,virustotal,yahoo +- python theHarvester.py -d metasploit.com -b baidu,bing,censys,crtsh,dnsdumpster,dogpile,duckduckgo,exalead,linkedin,netcraft,threatcrowd,trello,twitter,virustotal,yahoo -l 200 - pytest - flake8 . --count --show-source --statistics #- mypy *.py From d91fb6333e535194dc4dd6a016f41753b7435cdc Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Mon, 23 Sep 2019 12:01:26 -0400 Subject: [PATCH 09/12] Updated email parser to shift letters. --- theHarvester/parsers/myparser.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/theHarvester/parsers/myparser.py b/theHarvester/parsers/myparser.py index ddc03c94..b2fa519d 100644 --- a/theHarvester/parsers/myparser.py +++ b/theHarvester/parsers/myparser.py @@ -28,7 +28,10 @@ class Parser: reg_emails = re.compile(r'[a-zA-Z0-9.\-_+#~!$&\',;=:]+' + '@' + '[a-zA-Z0-9.-]*' + self.word.replace('www.', '')) self.temp = reg_emails.findall(self.results) emails = self.unique() - return emails + true_emails = {str(email)[1:].lower().strip() if len(str(email)) > 1 and str(email)[0] == '.' + else len(str(email)) > 1 and str(email).lower().strip() for email in emails} + # if email starts with dot shift email string and make sure all emails are lowercase + return true_emails def fileurls(self, file): urls = [] From cd34ec3ba9c9f47dd36581138d76c0ab3e999046 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Mon, 23 Sep 2019 21:00:53 -0400 Subject: [PATCH 10/12] Updated host checker to use asyncio and aiodns and changed how hosts are printed. --- theHarvester/__main__.py | 14 +++++-------- theHarvester/lib/hostchecker.py | 37 +++++++++++++++++++++++++-------- 2 files changed, 33 insertions(+), 18 deletions(-) diff --git a/theHarvester/__main__.py b/theHarvester/__main__.py index f0dc79b7..08799237 100644 --- a/theHarvester/__main__.py +++ b/theHarvester/__main__.py @@ -262,7 +262,7 @@ def start(): if isinstance(e, MissingKey): print(e) else: - print(e) + print(f'An exception has occurred in Intelx search: {e}') elif engineitem == 'linkedin': print('\033[94m[*] Searching Linkedin. \033[0m') @@ -436,20 +436,16 @@ def start(): full_host = hostchecker.Checker(all_hosts) full = full_host.check() for host in full: - ip = host.split(':')[1] - print(host) - if ip != 'empty': - if host_ip.count(ip.lower()): - pass - else: - host_ip.append(ip.lower()) + host = str(host) + print(host.lower()) db = stash.stash_manager() db.store_all(word, host_ip, 'ip', 'DNS-resolver') length_urls = len(trello_urls) if length_urls == 0: - print('\n[*] No Trello URLs found.') + if len(engines) >= 1 and 'trello' in engines: + print('\n[*] No Trello URLs found.') else: total = length_urls print('\n[*] Trello URLs found: ' + str(total)) diff --git a/theHarvester/lib/hostchecker.py b/theHarvester/lib/hostchecker.py index 0c3997b1..3c17c49d 100644 --- a/theHarvester/lib/hostchecker.py +++ b/theHarvester/lib/hostchecker.py @@ -2,24 +2,43 @@ # encoding: utf-8 """ Created by laramies on 2008-08-21. +Revised to use aiodns & asyncio on 2019-09-23 """ +import aiodns +import asyncio import socket class Checker: - def __init__(self, hosts): + def __init__(self, hosts: list): self.hosts = hosts self.realhosts = [] + @staticmethod + async def query(host, resolver) -> [list, str]: + try: + result = await resolver.gethostbyname(host, socket.AF_INET) + return result + except Exception as e: + # print(f'An error occurred in query: {e}') + return f"{host}:" + def check(self): - for x in self.hosts: - x = str(x) - try: - res = socket.gethostbyname(x) - res = str(res) - self.realhosts.append(x + ':' + res) - except Exception: - self.realhosts.append(x + ':' + 'empty') + loop = asyncio.get_event_loop() + resolver = aiodns.DNSResolver(loop=loop) + for host in self.hosts: + resp = self.query(host, resolver) + result = loop.run_until_complete(resp) + true_result = '' + if isinstance(result, str): + true_result = result + elif result != '' and not isinstance(result, str) and result.addresses is not None \ + and result.addresses != []: + result = result.addresses + result.sort() + true_result = f"{host}:{', '.join(map(str, result))}" + self.realhosts.append(true_result) + loop.close() return self.realhosts From a635b2b8dfd3a09dd63a0bf270106430e7c0adc6 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Mon, 23 Sep 2019 21:02:57 -0400 Subject: [PATCH 11/12] Added aiodns to requirements.txt --- requirements.txt | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/requirements.txt b/requirements.txt index dd796015..932382e2 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,3 +1,4 @@ +aiodns==2.0.0 beautifulsoup4==4.8.0 censys==0.0.8 dnspython==1.16.0 @@ -9,4 +10,4 @@ pytest==5.1.2 PyYaml==5.1.2 requests==2.22.0 shodan==1.15.0 -texttable==1.6.2 \ No newline at end of file +texttable==1.6.2 From fb44136d200b71dfcac7700f263c44aff1d11a25 Mon Sep 17 00:00:00 2001 From: NotoriousRebel Date: Mon, 23 Sep 2019 21:10:51 -0400 Subject: [PATCH 12/12] Updated line to fix pep8 error. --- theHarvester/lib/hostchecker.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/theHarvester/lib/hostchecker.py b/theHarvester/lib/hostchecker.py index 3c17c49d..733a0b04 100644 --- a/theHarvester/lib/hostchecker.py +++ b/theHarvester/lib/hostchecker.py @@ -21,7 +21,7 @@ class Checker: try: result = await resolver.gethostbyname(host, socket.AF_INET) return result - except Exception as e: + except Exception: # print(f'An error occurred in query: {e}') return f"{host}:"