diff --git a/.github/workflows/theHarvester.yml b/.github/workflows/theHarvester.yml index c006e4e9..eb29e14f 100644 --- a/.github/workflows/theHarvester.yml +++ b/.github/workflows/theHarvester.yml @@ -26,9 +26,57 @@ jobs: - name: Install dependencies run: | pip install -r requirements.txt - - name: Run theHarvester + - name: Run theHarvester module baidu run: | - python theHarvester.py -d metasploit.com -b baidu,bing,censys,crtsh,dnsdumpster,dogpile,duckduckgo,exalead,linkedin,netcraft,threatcrowd,trello,twitter,virustotal,yahoo + python theHarvester.py -d metasploit.com -b baidu + - name: Run theHarvester module bing + run: | + python theHarvester.py -d metasploit.com -b bing + - name: Run theHarvester module censys + run: | + python theHarvester.py -d metasploit.com -b censys + - name: Run theHarvester module crtsh + run: | + python theHarvester.py -d metasploit.com -b crtsh + - name: Run theHarvester module dnsdumpster + run: | + python theHarvester.py -d metasploit.com -b dnsdumpster + - name: Run theHarvester module dogplie + run: | + python theHarvester.py -d metasploit.com -b dogpile + - name: Run theHarvester module duckduckgo + run: | + python theHarvester.py -d metasploit.com -b duckduckgo + - name: Run theHarvester module exalead + run: | + python theHarvester.py -d metasploit.com -b exalead + - name: Run theHarvester module google + run: | + python theHarvester.py -d metasploit.com -b google + - name: Run theHarvester module linkedin + run: | + python theHarvester.py -d metasploit.com -b linkedin + - name: Run theHarvester module linkedin_links + run: | + python theHarvester.py -d metasploit.com -b linkedin_links + - name: Run theHarvester module netcraft + run: | + python theHarvester.py -d metasploit.com -b netcraft + - name: Run theHarvester module threatcrowd + run: | + python theHarvester.py -d metasploit.com -b threatcrowd + - name: Run theHarvester module trello + run: | + python theHarvester.py -d metasploit.com -b trello + - name: Run theHarvester module twitter + run: | + python theHarvester.py -d metasploit.com -b twitter + - name: Run theHarvester module virustotal + run: | + python theHarvester.py -d metasploit.com -b virustotal + - name: Run theHarvester module yahoo + run: | + python theHarvester.py -d metasploit.com -b yahoo - name: Lint with flake8 run: | # stop the build if there are Python syntax errors or undefined names diff --git a/.gitignore b/.gitignore index 83c17fa5..30769947 100644 --- a/.gitignore +++ b/.gitignore @@ -8,3 +8,5 @@ api-keys.yaml debug_results.txt tests/myparser.py venv +.mypy_cache +.pytest_cache diff --git a/.github/contributing.md b/CONTRIBUTING.md similarity index 67% rename from .github/contributing.md rename to CONTRIBUTING.md index f6f816d5..ef5fd561 100644 --- a/.github/contributing.md +++ b/CONTRIBUTING.md @@ -1,6 +1,5 @@ # Contributing to theHarvester Project Welcome to theHarvester project, so you would like to contribute. - The following below must be met to get accepted. # CI @@ -12,7 +11,9 @@ For new modules a unit test for that module is required and we use pytest. # Coding Standards * No single letter variables and variable names must represent the action that it is performing * Have static typing on functions etc +* Make sure no errors are reported from mypy +* No issues reported with flake8 # Submitting Bugs -If you have a bug in a module that you want to submit an issue for and know how to write python code. -Please create a unit test for that bug and submit a fix for it \ No newline at end of file +If you find a bug in a module that you want to submit an issue for and know how to write python code. +Please create a unit test for that bug(If possible) and submit a fix for it as it would be a big help to the project. \ No newline at end of file diff --git a/README.md b/README.md index c1631fee..91941510 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,5 @@ ![theHarvester](https://github.com/laramies/theHarvester/blob/master/theHarvester-logo.png) - [![Build Status](https://travis-ci.com/laramies/theHarvester.svg?branch=master)](https://travis-ci.com/laramies/theHarvester) [![Language grade: Python](https://img.shields.io/lgtm/grade/python/g/laramies/theHarvester.svg?logo=lgtm&logoWidth=18)](https://lgtm.com/projects/g/laramies/theHarvester/context:python) [![Rawsec's CyberSecurity Inventory](https://inventory.rawsec.ml/img/badges/Rawsec-inventoried-FF5050_flat_without_logo.svg)](https://inventory.rawsec.ml/) @@ -67,7 +66,7 @@ Active: ------- * DNS brute force: dictionary brute force enumeration * DNS reverse lookup: reverse lookup of IP´s discovered in order to find hostnames -* DNS TDL expansion: TLD dictionary brute force enumeration +* DNS TLD expansion: TLD dictionary brute force enumeration Modules that require an API key: -------------------------------- diff --git a/tests/discovery/test_linkedin_links.py b/tests/discovery/test_linkedin_links.py index 2dc6f3dc..b0c710c0 100644 --- a/tests/discovery/test_linkedin_links.py +++ b/tests/discovery/test_linkedin_links.py @@ -1,17 +1,43 @@ #!/usr/bin/env python3 # coding=utf-8 from theHarvester.discovery import linkedinsearch +from theHarvester.discovery.constants import splitter import pytest +import os +import re class TestGetLinks(object): + def test_splitter(self): + results = [ + 'https://www.linkedin.com/in/don-draper-b1045618', + 'https://www.linkedin.com/in/don-draper-b59210a', + 'https://www.linkedin.com/in/don-draper-b5bb50b3', + 'https://www.linkedin.com/in/don-draper-b83ba26', + 'https://www.linkedin.com/in/don-draper-b854a51' + ] + filtered_results = splitter(results) + assert len(filtered_results) == 1 + def test_get_links(self): search = linkedinsearch.SearchLinkedin("facebook.com", '100') search.process() links = search.get_links() - for link in links: - print(link) + assert type(links) == list + + def test_links_linkedin(self): + dir_path = os.path.dirname(os.path.realpath(__file__)) + mock_response = open(dir_path + "/test_linkedin_links.txt") + mock_response_content = mock_response.read() + mock_response.close() + reg_links = re.compile(r"url=https:\/\/www\.linkedin.com(.*?)&") + temp = reg_links.findall(mock_response_content) + resul = [] + for regex_item in temp: + stripped_url = regex_item.replace("url=", "") + resul.append("https://www.linkedin.com" + stripped_url) + assert set(resul) if __name__ == '__main__': diff --git a/tests/discovery/test_linkedin_links.txt b/tests/discovery/test_linkedin_links.txt new file mode 100644 index 00000000..b8804830 --- /dev/null +++ b/tests/discovery/test_linkedin_links.txt @@ -0,0 +1,5 @@ +LinkedIn
https://www.linkedin.com/in/gm-tuhin-ialam-546526b8
Albany, New York Area - Facebook Advertising
Gm Tuhin.ialam. facebook.com at Facebook Advertising. Albany, New York Area.
+Marketing and Advertising. Facebook Advertising. 0 connections ...

NIKUL www.facebook.com/nikulact - Modeling - Self Modeling ...

https://in.linkedin.com/in/nikulact
Ahmedabad Area, India - Self Modeling
View NIKUL www.facebook.com/nikulact's profile on LinkedIn, the world's largest
+professional community. NIKUL has 1 job listed on their profile. See the ...

Victor Scott - Metal Band facebook.com/alchemyoftime - Alchemy of ...

https://www.linkedin.com/in/victor-scott-9a967343
Albany, New York Area - Alchemy of Time
Victor Scott. Metal Band facebook.com/alchemyoftime at Alchemy of Time. Albany
+, New York Area. Music. Alchemy of Time. 1 connection ...

elkhorbat lkhorbat - http://www.facebook.com/pages/elkhorbat ...

https://www.linkedin.com/in/elkhorbat-lkhorbat-6028b33a
United States - http://www.facebook.com/pages/elkhorbat/302997479939
View elkhorbat lkhorbat's profile on LinkedIn, the world's largest professional
+community. elkhorbat has 1 job listed on their profile. See the complete profile on
diff --git a/theHarvester/__main__.py b/theHarvester/__main__.py index 533d9da6..9cf0b2fc 100644 --- a/theHarvester/__main__.py +++ b/theHarvester/__main__.py @@ -38,7 +38,6 @@ def start(): hunter, intelx, linkedin, linkedin_links, netcraft, securityTrails, threatcrowd, trello, twitter, vhost, virustotal, yahoo''') - parser.add_argument('-x', '--exclude', help='exclude options when using all sources', type=str) args = parser.parse_args() try: diff --git a/theHarvester/discovery/constants.py b/theHarvester/discovery/constants.py index 7549c4c6..0fbe8c91 100644 --- a/theHarvester/discovery/constants.py +++ b/theHarvester/discovery/constants.py @@ -4,6 +4,31 @@ import random googleUA = 'Mozilla/5.0 (Windows NT 6.2) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/28.0.1464.0 Safari/537.36' +def splitter(links): + """ + Method that tries to remove duplicates + LinkedinLists pulls a lot of profiles with the same name. + This method triest to remove duplicates from the list. + :param links: list of links to remove duplicates from + :return: unique-ish list + """ + unique_list = [] + name_check = [] + for url in links: + tail = url.split("/")[-1] + if len(tail) == 2 or tail == "zh-cn": + tail = url.split("/")[-2] + name = tail.split("-") + if len(name) > 1: + joined_name = name[0] + name[1] + else: + joined_name = name[0] + if joined_name not in name_check: + unique_list.append(url) + name_check.append(joined_name) + return unique_list + + def filter(lst): """ Method that filters list diff --git a/theHarvester/discovery/linkedinsearch.py b/theHarvester/discovery/linkedinsearch.py index 22a58699..ef774df3 100644 --- a/theHarvester/discovery/linkedinsearch.py +++ b/theHarvester/discovery/linkedinsearch.py @@ -36,7 +36,7 @@ class SearchLinkedin: def get_links(self): links = myparser.Parser(self.totalresults, self.word) - return links.links_linkedin() + return splitter(links.links_linkedin()) def process(self): while self.counter < self.limit: diff --git a/theHarvester/lib/markup.py b/theHarvester/lib/markup.py index c749b75a..7466ab99 100644 --- a/theHarvester/lib/markup.py +++ b/theHarvester/lib/markup.py @@ -1,16 +1,19 @@ -# This code is in the public domain, it comes with absolutely no -# warranty and you can do absolutely whatever you want with it. +# This code is in the public domain, it comes +# with absolutely no warranty and you can do +# absolutely whatever you want with it. +# type: ignore -__date__ = '17 May 2007' -__version__ = '1.7' +__date__ = '16 March 2015' +__version__ = '1.10' __doc__ = """ This is markup.py - a Python module that attempts to make it easier to generate HTML/XML from a Python program in an intuitive, lightweight, customizable and pythonic way. +It works with both python 2 and 3. The code is in the public domain. -Version: {0} as of {1}. +Version: %s as of %s. Documentation and further info is at http://markup.sourceforge.net/ @@ -18,37 +21,47 @@ Please send bug reports, feature requests, enhancement ideas or questions to nogradi at gmail dot com. Installation: drop markup.py somewhere into your Python path. -""".format(__version__, __date__) +""" % (__version__, __date__) + + +basestring = str +String = str +long = int + +# tags which are reserved python keywords will be referred +# to by a leading underscore otherwise we end up with a syntax error +import keyword class Element: - """This class handles the addition of a new element.""" def __init__(self, tag, case='lower', parent=None): self.parent = parent - if case == 'lower': - self.tag = tag.lower() - else: + if case == 'upper': self.tag = tag.upper() + elif case == 'lower': + self.tag = tag.lower() + elif case == 'given': + self.tag = tag + else: + self.tag = tag def __call__(self, *args, **kwargs): if len(args) > 1: raise ArgumentError(self.tag) - # If class_ was defined in parent, it should be added to every element. + # if class_ was defined in parent it should be added to every element if self.parent is not None and self.parent.class_ is not None: if 'class_' not in kwargs: kwargs['class_'] = self.parent.class_ if self.parent is None and len(args) == 1: - x = [self.render(self.tag, False, myarg, mydict) - for myarg, mydict in _argsdicts(args, kwargs)] + x = [self.render(self.tag, False, myarg, mydict) for myarg, mydict in _argsdicts(args, kwargs)] return '\n'.join(x) elif self.parent is None and len(args) == 0: - x = [self.render(self.tag, True, myarg, mydict) - for myarg, mydict in _argsdicts(args, kwargs)] + x = [self.render(self.tag, True, myarg, mydict) for myarg, mydict in _argsdicts(args, kwargs)] return '\n'.join(x) if self.tag in self.parent.twotags: @@ -57,8 +70,7 @@ class Element: elif self.tag in self.parent.onetags: if len(args) == 0: for myarg, mydict in _argsdicts(args, kwargs): - # Here myarg is always None, because len( args ) = 0. - self.render(self.tag, True, myarg, mydict) + self.render(self.tag, True, myarg, mydict) # here myarg is always None, because len( args ) = 0 else: raise ClosingError(self.tag) elif self.parent.mode == 'strict_html' and self.tag in self.parent.deptags: @@ -70,13 +82,10 @@ class Element: """Append the actual tags to content.""" out = "<%s" % tag - for key, value in kwargs.items(): - # When value is None, that means stuff like <... checked>. - if value is not None: - # Strip this so class_ will mean class, etc. - key = key.strip('_') - # Special cases, maybe change _ to - overall? - if key == 'http_equiv': + for key, value in list(kwargs.items()): + if value is not None: # when value is None that means stuff like <... checked> + key = key.strip('_') # strip this so class_ will mean class, etc. + if key == 'http_equiv': # special cases, maybe change _ to - overall? key = 'http-equiv' elif key == 'accept_charset': key = 'accept-charset' @@ -116,11 +125,7 @@ class Element: class Page: - """This is our main class representing a document. Elements are added as - attributes of an instance of this class.""" - - def __init__(self, mode='strict_html', case='lower', - onetags=None, twotags=None, separator='\n', class_=None): + def __init__(self, mode='strict_html', case='lower', onetags=None, twotags=None, separator='\n', class_=None): """Stuff that effects the whole document. mode -- 'strict_html' for HTML 4.01 (default) @@ -130,6 +135,7 @@ class Page: case -- 'lower' element names will be printed in lower case (default) 'upper' they will be printed in upper case + 'given' element names will be printed as they are given onetags -- list or tuple of valid elements with opening tags only twotags -- list or tuple of valid elements with both opening and closing tags @@ -141,36 +147,16 @@ class Page: class_ -- a class that will be added to every element if defined""" - valid_onetags = [ - "AREA", - "BASE", - "BR", - "COL", - "FRAME", - "HR", - "IMG", - "INPUT", - "LINK", - "META", - "PARAM"] - valid_twotags = [ - "A", "ABBR", "ACRONYM", "ADDRESS", "B", "BDO", "BIG", "BLOCKQUOTE", "BODY", "BUTTON", - "CAPTION", "CITE", "CODE", "COLGROUP", "DD", "DEL", "DFN", "DIV", "DL", "DT", "EM", "FIELDSET", - "FORM", "FRAMESET", "H1", "H2", "H3", "H4", "H5", "H6", "HEAD", "HTML", "I", "IFRAME", "INS", - "KBD", "LABEL", "LEGEND", "LI", "MAP", "NOFRAMES", "NOSCRIPT", "OBJECT", "OL", "OPTGROUP", - "OPTION", "P", "PRE", "Q", "SAMP", "SCRIPT", "SELECT", "SMALL", "SPAN", "STRONG", "STYLE", - "SUB", "SUP", "TABLE", "TBODY", "TD", "TEXTAREA", "TFOOT", "TH", "THEAD", "TITLE", "TR", - "TT", "UL", "VAR"] + valid_onetags = ["AREA", "BASE", "BR", "COL", "FRAME", "HR", "IMG", "INPUT", "LINK", "META", "PARAM"] + valid_twotags = ["A", "ABBR", "ACRONYM", "ADDRESS", "B", "BDO", "BIG", "BLOCKQUOTE", "BODY", "BUTTON", + "CAPTION", "CITE", "CODE", "COLGROUP", "DD", "DEL", "DFN", "DIV", "DL", "DT", "EM", "FIELDSET", + "FORM", "FRAMESET", "H1", "H2", "H3", "H4", "H5", "H6", "HEAD", "HTML", "I", "IFRAME", "INS", + "KBD", "LABEL", "LEGEND", "LI", "MAP", "NOFRAMES", "NOSCRIPT", "OBJECT", "OL", "OPTGROUP", + "OPTION", "P", "PRE", "Q", "SAMP", "SCRIPT", "SELECT", "SMALL", "SPAN", "STRONG", "STYLE", + "SUB", "SUP", "TABLE", "TBODY", "TD", "TEXTAREA", "TFOOT", "TH", "THEAD", "TITLE", "TR", + "TT", "UL", "VAR"] deprecated_onetags = ["BASEFONT", "ISINDEX"] - deprecated_twotags = [ - "APPLET", - "CENTER", - "DIR", - "FONT", - "MENU", - "S", - "STRIKE", - "U"] + deprecated_twotags = ["APPLET", "CENTER", "DIR", "FONT", "MENU", "S", "STRIKE", "U"] self.header = [] self.content = [] @@ -178,23 +164,23 @@ class Page: self.case = case self.separator = separator - # init( ) sets it to True so we know that has to be printed at the end. + # init( ) sets it to True so we know that has to be printed at the end self._full = False self.class_ = class_ if mode == 'strict_html' or mode == 'html': self.onetags = valid_onetags - self.onetags += [str.lower(x) for x in self.onetags] + self.onetags += list(map(String.lower, self.onetags)) self.twotags = valid_twotags - self.twotags += [str.lower(x) for x in self.twotags] + self.twotags += list(map(String.lower, self.twotags)) self.deptags = deprecated_onetags + deprecated_twotags - self.deptags += [str.lower(x) for x in self.deptags] + self.deptags += list(map(String.lower, self.deptags)) self.mode = 'strict_html' elif mode == 'loose_html': self.onetags = valid_onetags + deprecated_onetags - self.onetags += [str.lower(x) for x in self.onetags] + self.onetags += list(map(String.lower, self.onetags)) self.twotags = valid_twotags + deprecated_twotags - self.onetags += [str.lower(x) for x in self.twotags] + self.twotags += list(map(String.lower, self.twotags)) self.mode = mode elif mode == 'xml': if onetags and twotags: @@ -210,8 +196,16 @@ class Page: raise ModeError(mode) def __getattr__(self, attr): + + # tags should start with double underscore if attr.startswith("__") and attr.endswith("__"): raise AttributeError(attr) + # tag with single underscore should be a reserved keyword + if attr.startswith('_'): + attr = attr.lstrip('_') + if attr not in keyword.kwlist: + raise AttributeError(attr) + return Element(attr, case=self.case, parent=self) def __str__(self): @@ -252,7 +246,7 @@ class Page: self.content.append(text) def init(self, lang='en', css=None, metainfo=None, title=None, header=None, - footer=None, charset=None, encoding=None, doctype=None, bodyattrs=None, script=None): + footer=None, charset=None, encoding=None, doctype=None, bodyattrs=None, script=None, base=None): """This method is used for complete documents with appropriate doctype, encoding, title, etc information. For an HTML/XML snippet omit this method. @@ -267,11 +261,14 @@ class Page: into meta element(s) as (ignored in xml mode) + base -- set the tag in + bodyattrs --a dictionary in the form { 'key':'value', ... } which will be added as attributes of the element as (ignored in xml mode) script -- dictionary containing src:type pairs, + or a list of [ 'src1', 'src2', ... ] in which case 'javascript' is assumed for all title -- the title of the document as a string to be inserted into a title element as my title (ignored in xml mode) @@ -303,10 +300,7 @@ class Page: self.html(lang=lang) self.head() if charset is not None: - self.meta( - http_equiv='Content-Type', - content="text/html; charset=%s" % - charset) + self.meta(http_equiv='Content-Type', content="text/html; charset=%s" % charset) if metainfo is not None: self.metainfo(metainfo) if css is not None: @@ -315,6 +309,8 @@ class Page: self.title(title) if script is not None: self.scripts(script) + if base is not None: + self.base(href='%s' % base) self.head.close() if bodyattrs is not None: self.body(**bodyattrs) @@ -334,45 +330,40 @@ class Page: self.header.append(doctype) def css(self, filelist): - """This convenience function is only useful for html. It adds CSS - stylesheet(s) to the document via the element.""" + """This convenience function is only useful for html. + It adds css stylesheet(s) to the document via the element.""" - if isinstance(filelist, str): - self.link( - href=filelist, - rel='stylesheet', - type='text/css', - media='all') + if isinstance(filelist, basestring): + self.link(href=filelist, rel='stylesheet', type='text/css', media='all') else: for file in filelist: - self.link( - href=file, - rel='stylesheet', - type='text/css', - media='all') + self.link(href=file, rel='stylesheet', type='text/css', media='all') def metainfo(self, mydict): - """This convenience function is only useful for html. It adds meta - information via the element, the argument is a dictionary of - the form { 'name':'content' }.""" + """This convenience function is only useful for html. + It adds meta information via the element, the argument is + a dictionary of the form { 'name':'content' }.""" if isinstance(mydict, dict): - for name, content in mydict.items(): + for name, content in list(mydict.items()): self.meta(name=name, content=content) else: - raise TypeError( - "Metainfo should be called with a dictionary argument of name:content pairs.") + raise TypeError("Metainfo should be called with a dictionary argument of name:content pairs.") def scripts(self, mydict): - """Only useful in html, mydict is dictionary of src:type pairs will - be rendered as """ + """Only useful in html, mydict is dictionary of src:type pairs or a list + of script sources [ 'src1', 'src2', ... ] in which case 'javascript' is assumed for type. + Will be rendered as """ if isinstance(mydict, dict): - for src, type in mydict.items(): + for src, type in list(mydict.items()): self.script('', src=src, type='text/%s' % type) else: - raise TypeError( - "Script should be given a dictionary of src:type pairs.") + try: + for src in mydict: + self.script('', src=src, type='text/javascript') + except Exception: + raise TypeError("Script should be given a dictionary of src:type pairs or a list of javascript src's.") class _OneLiner: @@ -385,18 +376,26 @@ class _OneLiner: self.case = case def __getattr__(self, attr): + + # tags should start with double underscore if attr.startswith("__") and attr.endswith("__"): raise AttributeError(attr) + # tag with single underscore should be a reserved keyword + if attr.startswith('_'): + attr = attr.lstrip('_') + if attr not in keyword.kwlist: + raise AttributeError(attr) + return Element(attr, case=self.case, parent=None) oneliner = _OneLiner(case='lower') upper_oneliner = _OneLiner(case='upper') +given_oneliner = _OneLiner(case='given') def _argsdicts(args, mydict): - """A utility generator that pads argument list and dictionary values, will - only be called with len( args ) = 0, 1.""" + '''A utility generator that pads argument list and dictionary values, will only be called with len( args ) = 0, 1.''' if len(args) == 0: args = None, @@ -407,7 +406,9 @@ def _argsdicts(args, mydict): mykeys = list(mydict.keys()) myvalues = list(map(_totuple, list(mydict.values()))) + maxlength = max(list(map(len, [args] + myvalues))) + for i in range(maxlength): thisdict = {} for key, value in zip(mykeys, myvalues): @@ -424,11 +425,11 @@ def _argsdicts(args, mydict): def _totuple(x): - """Utility stuff to convert string, int, float, None or anything to a usable tuple.""" + """Utility stuff to convert string, int, long, float, None or anything to a usable tuple.""" - if isinstance(x, str): + if isinstance(x, basestring): out = x, - elif isinstance(x, (int, float)): + elif isinstance(x, (int, long, float)): out = str(x), elif x is None: out = None, @@ -441,7 +442,7 @@ def _totuple(x): def escape(text, newline=False): """Escape special html characters.""" - if isinstance(text, str): + if isinstance(text, basestring): if '&' in text: text = text.replace('&', '&') if '>' in text: @@ -465,7 +466,7 @@ _escape = escape def unescape(text): """Inverse of escape.""" - if isinstance(text, str): + if isinstance(text, basestring): if '&' in text: text = text.replace('&', '&') if '>' in text: @@ -479,19 +480,17 @@ def unescape(text): class Dummy: - """A dummy class for attaching attributes.""" pass doctype = Dummy() -doctype.frameset = "" -doctype.strict = "" -doctype.loose = "" +doctype.frameset = """""" +doctype.strict = """""" +doctype.loose = """""" class Russell: - """A dummy class that contains anything.""" def __contains__(self, item): @@ -499,7 +498,6 @@ class Russell: class MarkupError(Exception): - """All our exceptions subclass this.""" def __str__(self): @@ -507,48 +505,41 @@ class MarkupError(Exception): class ClosingError(MarkupError): - def __init__(self, tag): - self.message = "The element '{}' does not accept non-keyword arguments (has no closing tag)".format(tag) + self.message = "The element '%s' does not accept non-keyword arguments (has no closing tag)." % tag class OpeningError(MarkupError): - def __init__(self, tag): - self.message = "The element '{}' can not be opened.".format(tag) + self.message = "The element '%s' can not be opened." % tag class ArgumentError(MarkupError): - def __init__(self, tag): - self.message = "The element '{}' was called with more than one non-keyword argument.".format(tag) + self.message = "The element '%s' was called with more than one non-keyword argument." % tag class InvalidElementError(MarkupError): - def __init__(self, tag, mode): - self.message = "The element '{0}' is not valid for your mode '{1}'.".format( - tag, - mode) + self.message = "The element '%s' is not valid for your mode '%s'." % (tag, mode) class DeprecationError(MarkupError): - def __init__(self, tag): - self.message = "The element '{0}' is deprecated, instantiate markup.page with mode='loose_html' to allow it.".format(tag) + self.message = "The element '%s' is deprecated, instantiate markup.page with mode='loose_html' to allow it." % tag class ModeError(MarkupError): - def __init__(self, mode): - self.message = "Mode '{}' is invalid, possible values: strict_html, loose_html, xml.".format(mode) + self.message = "Mode '%s' is invalid, possible values: strict_html, html (alias for strict_html), loose_html, xml." % mode class CustomizationError(MarkupError): - def __init__(self): self.message = "If you customize the allowed elements, you must define both types 'onetags' and 'twotags'." if __name__ == '__main__': - print(__doc__) + import sys + + sys.stdout.write(__doc__) diff --git a/theHarvester/parsers/intelxparser.py b/theHarvester/parsers/intelxparser.py index 8c12b969..91c34aa7 100644 --- a/theHarvester/parsers/intelxparser.py +++ b/theHarvester/parsers/intelxparser.py @@ -4,7 +4,7 @@ class Parser: self.emails = set() self.hosts = set() - def parse_dictionaries(self, results): + def parse_dictionaries(self, results: dict) -> tuple: """ Parse method to parse json results :param results: Dictionary containing a list of dictionaries known as selectors diff --git a/theHarvester/parsers/myparser.py b/theHarvester/parsers/myparser.py index 09c8f825..83a843c1 100644 --- a/theHarvester/parsers/myparser.py +++ b/theHarvester/parsers/myparser.py @@ -81,10 +81,10 @@ class Parser: reg_links = re.compile(r"url=https:\/\/www\.linkedin.com(.*?)&") self.temp = reg_links.findall(self.results) resul = [] - for x in self.temp: - y = x.replace("url=", "") - resul.append("https://www.linkedin.com" + y) - return set(resul) + for regex in self.temp: + final_url = regex.replace("url=", "") + resul.append("https://www.linkedin.com" + final_url) + return resul def people_linkedin(self): reg_people = re.compile(r'">[a-zA-Z0-9._ -]* \| LinkedIn')