mirror of
https://github.com/laramies/theHarvester.git
synced 2026-10-01 05:55:00 +02:00
Merge remote-tracking branch 'origin/dev' into dev
This commit is contained in:
+10
-5
@@ -1,15 +1,20 @@
|
||||
dist: bionic
|
||||
language: python
|
||||
python:
|
||||
- '3.6'
|
||||
- '3.7'
|
||||
- '3.8-dev'
|
||||
cache: pip
|
||||
matrix:
|
||||
include:
|
||||
- python: '3.6'
|
||||
env: TEST_SUITE=suite_3_6
|
||||
- python: '3.7'
|
||||
env: TEST_SUITE=suite_3_7
|
||||
- python: '3.8-dev'
|
||||
env: TEST_SUITE=suite_3_8_dev
|
||||
before_install:
|
||||
- pip install -r requirements.txt
|
||||
install:
|
||||
- python setup.py test
|
||||
script:
|
||||
- python theHarvester.py -d metasploit.com -b baidu,bing,censys,crtsh,dnsdumpster,dogpile,duckduckgo,exalead,linkedin,netcraft,otx,threatcrowd,trello,twitter,virustotal,yahoo
|
||||
- python theHarvester.py -d metasploit.com -b baidu,bing,censys,crtsh,dnsdumpster,dogpile,duckduckgo,exalead,linkedin,netcraft,threatcrowd,trello,twitter,virustotal,yahoo -l 200
|
||||
- pytest
|
||||
- flake8 . --count --show-source --statistics
|
||||
#- mypy *.py
|
||||
|
||||
+2
-1
@@ -1,3 +1,4 @@
|
||||
aiodns==2.0.0
|
||||
beautifulsoup4==4.8.0
|
||||
censys==0.0.8
|
||||
dnspython==1.16.0
|
||||
@@ -9,4 +10,4 @@ pytest==5.1.3
|
||||
PyYaml==5.1.2
|
||||
requests==2.22.0
|
||||
shodan==1.17.0
|
||||
texttable==1.6.2
|
||||
texttable==1.6.2
|
||||
|
||||
+20
-26
@@ -61,14 +61,14 @@ def start():
|
||||
shodan = args.shodan
|
||||
start = args.start # type: int
|
||||
takeover_check = False
|
||||
trello_info = ([], False)
|
||||
trello_urls = []
|
||||
vhost = []
|
||||
virtual = args.virtual_host
|
||||
word = args.domain # type: str
|
||||
|
||||
if args.source is not None:
|
||||
engines = set(map(str.strip, args.source.split(',')))
|
||||
|
||||
engines = sorted(set(map(str.strip, args.source.split(','))))
|
||||
# Iterate through search engines in order
|
||||
if set(engines).issubset(Core.get_supportedengines()):
|
||||
print(f'\033[94m[*] Target: {word} \n \033[0m')
|
||||
|
||||
@@ -262,7 +262,7 @@ def start():
|
||||
if isinstance(e, MissingKey):
|
||||
print(e)
|
||||
else:
|
||||
print(e)
|
||||
print(f'An exception has occurred in Intelx search: {e}')
|
||||
|
||||
elif engineitem == 'linkedin':
|
||||
print('\033[94m[*] Searching Linkedin. \033[0m')
|
||||
@@ -361,13 +361,12 @@ def start():
|
||||
print('\033[94m[*] Searching Trello. \033[0m')
|
||||
from theHarvester.discovery import trello
|
||||
# Import locally or won't work.
|
||||
trello_search = trello.SearchTrello(word, limit)
|
||||
trello_search = trello.SearchTrello(word)
|
||||
trello_search.process()
|
||||
emails = filter(trello_search.get_emails())
|
||||
emails, hosts, urls = trello_search.get_results()
|
||||
all_emails.extend(emails)
|
||||
info = trello_search.get_urls()
|
||||
hosts = filter(info[0])
|
||||
trello_info = (info[1], True)
|
||||
hosts = filter(hosts)
|
||||
trello_urls = filter(urls)
|
||||
all_hosts.extend(hosts)
|
||||
db = stash.stash_manager()
|
||||
db.store_all(word, hosts, 'host', 'trello')
|
||||
@@ -453,27 +452,22 @@ def start():
|
||||
full_host = hostchecker.Checker(all_hosts)
|
||||
full = full_host.check()
|
||||
for host in full:
|
||||
ip = host.split(':')[1]
|
||||
print(host)
|
||||
if ip != 'empty':
|
||||
if host_ip.count(ip.lower()):
|
||||
pass
|
||||
else:
|
||||
host_ip.append(ip.lower())
|
||||
host = str(host)
|
||||
print(host.lower())
|
||||
|
||||
db = stash.stash_manager()
|
||||
db.store_all(word, host_ip, 'ip', 'DNS-resolver')
|
||||
|
||||
if trello_info[1] is True:
|
||||
trello_urls = trello_info[0]
|
||||
if trello_urls is []:
|
||||
print('\n[*] No URLs found.')
|
||||
else:
|
||||
total = len(trello_urls)
|
||||
print('\n[*] URLs found: ' + str(total))
|
||||
print('--------------------')
|
||||
for url in sorted(list(set(trello_urls))):
|
||||
print(url)
|
||||
length_urls = len(trello_urls)
|
||||
if length_urls == 0:
|
||||
if len(engines) >= 1 and 'trello' in engines:
|
||||
print('\n[*] No Trello URLs found.')
|
||||
else:
|
||||
total = length_urls
|
||||
print('\n[*] Trello URLs found: ' + str(total))
|
||||
print('--------------------')
|
||||
for url in sorted(trello_urls):
|
||||
print(url)
|
||||
|
||||
# DNS brute force
|
||||
# dnsres = []
|
||||
|
||||
@@ -1,30 +1,45 @@
|
||||
from theHarvester.discovery.constants import *
|
||||
from theHarvester.parsers import myparser
|
||||
import grequests
|
||||
import requests
|
||||
import random
|
||||
import time
|
||||
|
||||
|
||||
class SearchTrello:
|
||||
|
||||
def __init__(self, word, limit):
|
||||
def __init__(self, word):
|
||||
self.word = word.replace(' ', '%20')
|
||||
self.results = ""
|
||||
self.totalresults = ""
|
||||
self.server = 'www.google.com'
|
||||
self.quantity = '100'
|
||||
self.limit = limit
|
||||
self.limit = 300
|
||||
self.trello_urls = []
|
||||
self.hostnames = []
|
||||
self.counter = 0
|
||||
|
||||
def do_search(self):
|
||||
base_url = f'https://{self.server}/search?num=100&start=xx&hl=en&q=site%3Atrello.com%20{self.word}'
|
||||
base_url = f'https://{self.server}/search?num=300&start=xx&hl=en&q=site%3Atrello.com%20{self.word}'
|
||||
urls = [base_url.replace("xx", str(num)) for num in range(0, self.limit, 20) if num <= self.limit]
|
||||
# limit is 20 as that is the most results google will show per num
|
||||
headers = {'User-Agent': googleUA}
|
||||
try:
|
||||
urls = [base_url.replace("xx", str(num)) for num in range(0, self.limit, 10) if num <= self.limit]
|
||||
request = (grequests.get(url, headers=headers) for url in urls)
|
||||
response = grequests.imap(request, size=5)
|
||||
for entry in response:
|
||||
self.totalresults += entry.content.decode('UTF-8')
|
||||
except Exception as e:
|
||||
print(e)
|
||||
for url in urls:
|
||||
try:
|
||||
resp = requests.get(url, headers=headers)
|
||||
self.results = resp.text
|
||||
if search(self.results):
|
||||
try:
|
||||
self.results = google_workaround(base_url)
|
||||
if isinstance(self.results, bool):
|
||||
print('Google is blocking your ip and the workaround, returning')
|
||||
return
|
||||
except Exception as e:
|
||||
print(e)
|
||||
self.totalresults += self.results
|
||||
time.sleep(getDelay() - .5)
|
||||
except Exception as e:
|
||||
print(f'An exception has occurred in trello: {e}')
|
||||
|
||||
def get_emails(self):
|
||||
rawres = myparser.Parser(self.totalresults, self.word)
|
||||
@@ -33,17 +48,18 @@ class SearchTrello:
|
||||
def get_urls(self):
|
||||
try:
|
||||
rawres = myparser.Parser(self.totalresults, 'trello.com')
|
||||
trello_urls = rawres.urls()
|
||||
visited = set()
|
||||
for url in trello_urls:
|
||||
# Iterate through Trello URLs gathered and visit them, append text to totalresults.
|
||||
if url not in visited: # Make sure visiting unique URLs.
|
||||
visited.add(url)
|
||||
request = grequests.get(url=url, headers={'User-Agent': googleUA})
|
||||
response = grequests.map([request])
|
||||
self.totalresults = response[0].content.decode('UTF-8')
|
||||
self.trello_urls = set(rawres.urls())
|
||||
self.totalresults = ''
|
||||
# reset what totalresults as before it was just google results now it is trello results
|
||||
headers = {'User-Agent': random.choice(['curl/7.37.0', 'Wget/1.19.4'])}
|
||||
# do not change the headers
|
||||
req = (grequests.get(url, headers=headers, timeout=4) for url in self.trello_urls)
|
||||
responses = grequests.imap(req, size=8)
|
||||
for response in responses:
|
||||
self.totalresults += response.content.decode('UTF-8')
|
||||
|
||||
rawres = myparser.Parser(self.totalresults, self.word)
|
||||
return rawres.hostnames(), trello_urls
|
||||
self.hostnames = rawres.hostnames()
|
||||
except Exception as e:
|
||||
print(f'Error occurred: {e}')
|
||||
|
||||
@@ -51,3 +67,6 @@ class SearchTrello:
|
||||
self.do_search()
|
||||
self.get_urls()
|
||||
print(f'\tSearching {self.counter} results.')
|
||||
|
||||
def get_results(self) -> tuple:
|
||||
return self.get_emails(), self.hostnames, self.trello_urls
|
||||
|
||||
@@ -2,24 +2,43 @@
|
||||
# encoding: utf-8
|
||||
"""
|
||||
Created by laramies on 2008-08-21.
|
||||
Revised to use aiodns & asyncio on 2019-09-23
|
||||
"""
|
||||
|
||||
import aiodns
|
||||
import asyncio
|
||||
import socket
|
||||
|
||||
|
||||
class Checker:
|
||||
|
||||
def __init__(self, hosts):
|
||||
def __init__(self, hosts: list):
|
||||
self.hosts = hosts
|
||||
self.realhosts = []
|
||||
|
||||
@staticmethod
|
||||
async def query(host, resolver) -> [list, str]:
|
||||
try:
|
||||
result = await resolver.gethostbyname(host, socket.AF_INET)
|
||||
return result
|
||||
except Exception:
|
||||
# print(f'An error occurred in query: {e}')
|
||||
return f"{host}:"
|
||||
|
||||
def check(self):
|
||||
for x in self.hosts:
|
||||
x = str(x)
|
||||
try:
|
||||
res = socket.gethostbyname(x)
|
||||
res = str(res)
|
||||
self.realhosts.append(x + ':' + res)
|
||||
except Exception:
|
||||
self.realhosts.append(x + ':' + 'empty')
|
||||
loop = asyncio.get_event_loop()
|
||||
resolver = aiodns.DNSResolver(loop=loop)
|
||||
for host in self.hosts:
|
||||
resp = self.query(host, resolver)
|
||||
result = loop.run_until_complete(resp)
|
||||
true_result = ''
|
||||
if isinstance(result, str):
|
||||
true_result = result
|
||||
elif result != '' and not isinstance(result, str) and result.addresses is not None \
|
||||
and result.addresses != []:
|
||||
result = result.addresses
|
||||
result.sort()
|
||||
true_result = f"{host}:{', '.join(map(str, result))}"
|
||||
self.realhosts.append(true_result)
|
||||
loop.close()
|
||||
return self.realhosts
|
||||
|
||||
@@ -25,10 +25,13 @@ class Parser:
|
||||
self.genericClean()
|
||||
# Local part is required, charset is flexible.
|
||||
# https://tools.ietf.org/html/rfc6531 (removed * and () as they provide FP mostly)
|
||||
reg_emails = re.compile(r'[a-zA-Z0-9.\-_+#~!$&\',;=:]+' + '@' + '[a-zA-Z0-9.-]*' + self.word)
|
||||
reg_emails = re.compile(r'[a-zA-Z0-9.\-_+#~!$&\',;=:]+' + '@' + '[a-zA-Z0-9.-]*' + self.word.replace('www.', ''))
|
||||
self.temp = reg_emails.findall(self.results)
|
||||
emails = self.unique()
|
||||
return emails
|
||||
true_emails = {str(email)[1:].lower().strip() if len(str(email)) > 1 and str(email)[0] == '.'
|
||||
else len(str(email)) > 1 and str(email).lower().strip() for email in emails}
|
||||
# if email starts with dot shift email string and make sure all emails are lowercase
|
||||
return true_emails
|
||||
|
||||
def fileurls(self, file):
|
||||
urls = []
|
||||
@@ -47,7 +50,10 @@ class Parser:
|
||||
reg_hosts = re.compile(r'[a-zA-Z0-9.-]*\.' + self.word)
|
||||
self.temp = reg_hosts.findall(self.results)
|
||||
hostnames = self.unique()
|
||||
return hostnames
|
||||
reg_hosts = re.compile(r'[a-zA-Z0-9.-]*\.' + self.word.replace('www.', ''))
|
||||
self.temp = reg_hosts.findall(self.results)
|
||||
hostnames.extend(self.unique())
|
||||
return list(set(hostnames))
|
||||
|
||||
def people_googleplus(self):
|
||||
self.results = re.sub('</b>', '', self.results)
|
||||
@@ -138,13 +144,11 @@ class Parser:
|
||||
return sets
|
||||
|
||||
def urls(self):
|
||||
found = re.finditer(r'https://(www\.)?trello.com/([a-zA-Z0-9\-_\.]+/?)*', self.results)
|
||||
for x in found:
|
||||
self.temp.append(x.group())
|
||||
urls = self.unique()
|
||||
found = re.finditer(r'(http|https)://(www\.)?trello.com/([a-zA-Z0-9\-_\.]+/?)*', self.results)
|
||||
urls = {match.group().strip() for match in found}
|
||||
return urls
|
||||
|
||||
def unique(self):
|
||||
def unique(self) -> list:
|
||||
self.new = []
|
||||
for x in self.temp:
|
||||
if x not in self.new:
|
||||
|
||||
Reference in New Issue
Block a user