Merge remote-tracking branch 'origin/dev' into dev

This commit is contained in:
Jay Townsend
2019-09-25 00:35:33 +01:00
6 changed files with 112 additions and 70 deletions
+10 -5
View File
@@ -1,15 +1,20 @@
dist: bionic
language: python
python:
- '3.6'
- '3.7'
- '3.8-dev'
cache: pip
matrix:
include:
- python: '3.6'
env: TEST_SUITE=suite_3_6
- python: '3.7'
env: TEST_SUITE=suite_3_7
- python: '3.8-dev'
env: TEST_SUITE=suite_3_8_dev
before_install:
- pip install -r requirements.txt
install:
- python setup.py test
script:
- python theHarvester.py -d metasploit.com -b baidu,bing,censys,crtsh,dnsdumpster,dogpile,duckduckgo,exalead,linkedin,netcraft,otx,threatcrowd,trello,twitter,virustotal,yahoo
- python theHarvester.py -d metasploit.com -b baidu,bing,censys,crtsh,dnsdumpster,dogpile,duckduckgo,exalead,linkedin,netcraft,threatcrowd,trello,twitter,virustotal,yahoo -l 200
- pytest
- flake8 . --count --show-source --statistics
#- mypy *.py
+2 -1
View File
@@ -1,3 +1,4 @@
aiodns==2.0.0
beautifulsoup4==4.8.0
censys==0.0.8
dnspython==1.16.0
@@ -9,4 +10,4 @@ pytest==5.1.3
PyYaml==5.1.2
requests==2.22.0
shodan==1.17.0
texttable==1.6.2
texttable==1.6.2
+20 -26
View File
@@ -61,14 +61,14 @@ def start():
shodan = args.shodan
start = args.start # type: int
takeover_check = False
trello_info = ([], False)
trello_urls = []
vhost = []
virtual = args.virtual_host
word = args.domain # type: str
if args.source is not None:
engines = set(map(str.strip, args.source.split(',')))
engines = sorted(set(map(str.strip, args.source.split(','))))
# Iterate through search engines in order
if set(engines).issubset(Core.get_supportedengines()):
print(f'\033[94m[*] Target: {word} \n \033[0m')
@@ -262,7 +262,7 @@ def start():
if isinstance(e, MissingKey):
print(e)
else:
print(e)
print(f'An exception has occurred in Intelx search: {e}')
elif engineitem == 'linkedin':
print('\033[94m[*] Searching Linkedin. \033[0m')
@@ -361,13 +361,12 @@ def start():
print('\033[94m[*] Searching Trello. \033[0m')
from theHarvester.discovery import trello
# Import locally or won't work.
trello_search = trello.SearchTrello(word, limit)
trello_search = trello.SearchTrello(word)
trello_search.process()
emails = filter(trello_search.get_emails())
emails, hosts, urls = trello_search.get_results()
all_emails.extend(emails)
info = trello_search.get_urls()
hosts = filter(info[0])
trello_info = (info[1], True)
hosts = filter(hosts)
trello_urls = filter(urls)
all_hosts.extend(hosts)
db = stash.stash_manager()
db.store_all(word, hosts, 'host', 'trello')
@@ -453,27 +452,22 @@ def start():
full_host = hostchecker.Checker(all_hosts)
full = full_host.check()
for host in full:
ip = host.split(':')[1]
print(host)
if ip != 'empty':
if host_ip.count(ip.lower()):
pass
else:
host_ip.append(ip.lower())
host = str(host)
print(host.lower())
db = stash.stash_manager()
db.store_all(word, host_ip, 'ip', 'DNS-resolver')
if trello_info[1] is True:
trello_urls = trello_info[0]
if trello_urls is []:
print('\n[*] No URLs found.')
else:
total = len(trello_urls)
print('\n[*] URLs found: ' + str(total))
print('--------------------')
for url in sorted(list(set(trello_urls))):
print(url)
length_urls = len(trello_urls)
if length_urls == 0:
if len(engines) >= 1 and 'trello' in engines:
print('\n[*] No Trello URLs found.')
else:
total = length_urls
print('\n[*] Trello URLs found: ' + str(total))
print('--------------------')
for url in sorted(trello_urls):
print(url)
# DNS brute force
# dnsres = []
+40 -21
View File
@@ -1,30 +1,45 @@
from theHarvester.discovery.constants import *
from theHarvester.parsers import myparser
import grequests
import requests
import random
import time
class SearchTrello:
def __init__(self, word, limit):
def __init__(self, word):
self.word = word.replace(' ', '%20')
self.results = ""
self.totalresults = ""
self.server = 'www.google.com'
self.quantity = '100'
self.limit = limit
self.limit = 300
self.trello_urls = []
self.hostnames = []
self.counter = 0
def do_search(self):
base_url = f'https://{self.server}/search?num=100&start=xx&hl=en&q=site%3Atrello.com%20{self.word}'
base_url = f'https://{self.server}/search?num=300&start=xx&hl=en&q=site%3Atrello.com%20{self.word}'
urls = [base_url.replace("xx", str(num)) for num in range(0, self.limit, 20) if num <= self.limit]
# limit is 20 as that is the most results google will show per num
headers = {'User-Agent': googleUA}
try:
urls = [base_url.replace("xx", str(num)) for num in range(0, self.limit, 10) if num <= self.limit]
request = (grequests.get(url, headers=headers) for url in urls)
response = grequests.imap(request, size=5)
for entry in response:
self.totalresults += entry.content.decode('UTF-8')
except Exception as e:
print(e)
for url in urls:
try:
resp = requests.get(url, headers=headers)
self.results = resp.text
if search(self.results):
try:
self.results = google_workaround(base_url)
if isinstance(self.results, bool):
print('Google is blocking your ip and the workaround, returning')
return
except Exception as e:
print(e)
self.totalresults += self.results
time.sleep(getDelay() - .5)
except Exception as e:
print(f'An exception has occurred in trello: {e}')
def get_emails(self):
rawres = myparser.Parser(self.totalresults, self.word)
@@ -33,17 +48,18 @@ class SearchTrello:
def get_urls(self):
try:
rawres = myparser.Parser(self.totalresults, 'trello.com')
trello_urls = rawres.urls()
visited = set()
for url in trello_urls:
# Iterate through Trello URLs gathered and visit them, append text to totalresults.
if url not in visited: # Make sure visiting unique URLs.
visited.add(url)
request = grequests.get(url=url, headers={'User-Agent': googleUA})
response = grequests.map([request])
self.totalresults = response[0].content.decode('UTF-8')
self.trello_urls = set(rawres.urls())
self.totalresults = ''
# reset what totalresults as before it was just google results now it is trello results
headers = {'User-Agent': random.choice(['curl/7.37.0', 'Wget/1.19.4'])}
# do not change the headers
req = (grequests.get(url, headers=headers, timeout=4) for url in self.trello_urls)
responses = grequests.imap(req, size=8)
for response in responses:
self.totalresults += response.content.decode('UTF-8')
rawres = myparser.Parser(self.totalresults, self.word)
return rawres.hostnames(), trello_urls
self.hostnames = rawres.hostnames()
except Exception as e:
print(f'Error occurred: {e}')
@@ -51,3 +67,6 @@ class SearchTrello:
self.do_search()
self.get_urls()
print(f'\tSearching {self.counter} results.')
def get_results(self) -> tuple:
return self.get_emails(), self.hostnames, self.trello_urls
+28 -9
View File
@@ -2,24 +2,43 @@
# encoding: utf-8
"""
Created by laramies on 2008-08-21.
Revised to use aiodns & asyncio on 2019-09-23
"""
import aiodns
import asyncio
import socket
class Checker:
def __init__(self, hosts):
def __init__(self, hosts: list):
self.hosts = hosts
self.realhosts = []
@staticmethod
async def query(host, resolver) -> [list, str]:
try:
result = await resolver.gethostbyname(host, socket.AF_INET)
return result
except Exception:
# print(f'An error occurred in query: {e}')
return f"{host}:"
def check(self):
for x in self.hosts:
x = str(x)
try:
res = socket.gethostbyname(x)
res = str(res)
self.realhosts.append(x + ':' + res)
except Exception:
self.realhosts.append(x + ':' + 'empty')
loop = asyncio.get_event_loop()
resolver = aiodns.DNSResolver(loop=loop)
for host in self.hosts:
resp = self.query(host, resolver)
result = loop.run_until_complete(resp)
true_result = ''
if isinstance(result, str):
true_result = result
elif result != '' and not isinstance(result, str) and result.addresses is not None \
and result.addresses != []:
result = result.addresses
result.sort()
true_result = f"{host}:{', '.join(map(str, result))}"
self.realhosts.append(true_result)
loop.close()
return self.realhosts
+12 -8
View File
@@ -25,10 +25,13 @@ class Parser:
self.genericClean()
# Local part is required, charset is flexible.
# https://tools.ietf.org/html/rfc6531 (removed * and () as they provide FP mostly)
reg_emails = re.compile(r'[a-zA-Z0-9.\-_+#~!$&\',;=:]+' + '@' + '[a-zA-Z0-9.-]*' + self.word)
reg_emails = re.compile(r'[a-zA-Z0-9.\-_+#~!$&\',;=:]+' + '@' + '[a-zA-Z0-9.-]*' + self.word.replace('www.', ''))
self.temp = reg_emails.findall(self.results)
emails = self.unique()
return emails
true_emails = {str(email)[1:].lower().strip() if len(str(email)) > 1 and str(email)[0] == '.'
else len(str(email)) > 1 and str(email).lower().strip() for email in emails}
# if email starts with dot shift email string and make sure all emails are lowercase
return true_emails
def fileurls(self, file):
urls = []
@@ -47,7 +50,10 @@ class Parser:
reg_hosts = re.compile(r'[a-zA-Z0-9.-]*\.' + self.word)
self.temp = reg_hosts.findall(self.results)
hostnames = self.unique()
return hostnames
reg_hosts = re.compile(r'[a-zA-Z0-9.-]*\.' + self.word.replace('www.', ''))
self.temp = reg_hosts.findall(self.results)
hostnames.extend(self.unique())
return list(set(hostnames))
def people_googleplus(self):
self.results = re.sub('</b>', '', self.results)
@@ -138,13 +144,11 @@ class Parser:
return sets
def urls(self):
found = re.finditer(r'https://(www\.)?trello.com/([a-zA-Z0-9\-_\.]+/?)*', self.results)
for x in found:
self.temp.append(x.group())
urls = self.unique()
found = re.finditer(r'(http|https)://(www\.)?trello.com/([a-zA-Z0-9\-_\.]+/?)*', self.results)
urls = {match.group().strip() for match in found}
return urls
def unique(self):
def unique(self) -> list:
self.new = []
for x in self.temp:
if x not in self.new: