mirror of
https://github.com/laramies/theHarvester.git
synced 2026-08-17 19:35:40 +02:00
Merge branch 'master' of https://github.com/laramies/theHarvester
This commit is contained in:
@@ -26,9 +26,57 @@ jobs:
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
pip install -r requirements.txt
|
||||
- name: Run theHarvester
|
||||
- name: Run theHarvester module baidu
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b baidu,bing,censys,crtsh,dnsdumpster,dogpile,duckduckgo,exalead,linkedin,netcraft,threatcrowd,trello,twitter,virustotal,yahoo
|
||||
python theHarvester.py -d metasploit.com -b baidu
|
||||
- name: Run theHarvester module bing
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b bing
|
||||
- name: Run theHarvester module censys
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b censys
|
||||
- name: Run theHarvester module crtsh
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b crtsh
|
||||
- name: Run theHarvester module dnsdumpster
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b dnsdumpster
|
||||
- name: Run theHarvester module dogplie
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b dogpile
|
||||
- name: Run theHarvester module duckduckgo
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b duckduckgo
|
||||
- name: Run theHarvester module exalead
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b exalead
|
||||
- name: Run theHarvester module google
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b google
|
||||
- name: Run theHarvester module linkedin
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b linkedin
|
||||
- name: Run theHarvester module linkedin_links
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b linkedin_links
|
||||
- name: Run theHarvester module netcraft
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b netcraft
|
||||
- name: Run theHarvester module threatcrowd
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b threatcrowd
|
||||
- name: Run theHarvester module trello
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b trello
|
||||
- name: Run theHarvester module twitter
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b twitter
|
||||
- name: Run theHarvester module virustotal
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b virustotal
|
||||
- name: Run theHarvester module yahoo
|
||||
run: |
|
||||
python theHarvester.py -d metasploit.com -b yahoo
|
||||
- name: Lint with flake8
|
||||
run: |
|
||||
# stop the build if there are Python syntax errors or undefined names
|
||||
|
||||
@@ -8,3 +8,5 @@ api-keys.yaml
|
||||
debug_results.txt
|
||||
tests/myparser.py
|
||||
venv
|
||||
.mypy_cache
|
||||
.pytest_cache
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
# Contributing to theHarvester Project
|
||||
Welcome to theHarvester project, so you would like to contribute.
|
||||
|
||||
The following below must be met to get accepted.
|
||||
|
||||
# CI
|
||||
@@ -12,7 +11,9 @@ For new modules a unit test for that module is required and we use pytest.
|
||||
# Coding Standards
|
||||
* No single letter variables and variable names must represent the action that it is performing
|
||||
* Have static typing on functions etc
|
||||
* Make sure no errors are reported from mypy
|
||||
* No issues reported with flake8
|
||||
|
||||
# Submitting Bugs
|
||||
If you have a bug in a module that you want to submit an issue for and know how to write python code.
|
||||
Please create a unit test for that bug and submit a fix for it
|
||||
If you find a bug in a module that you want to submit an issue for and know how to write python code.
|
||||
Please create a unit test for that bug(If possible) and submit a fix for it as it would be a big help to the project.
|
||||
@@ -1,6 +1,5 @@
|
||||

|
||||
|
||||
|
||||
[](https://travis-ci.com/laramies/theHarvester) [](https://lgtm.com/projects/g/laramies/theHarvester/context:python)
|
||||
[](https://inventory.rawsec.ml/)
|
||||
|
||||
@@ -67,7 +66,7 @@ Active:
|
||||
-------
|
||||
* DNS brute force: dictionary brute force enumeration
|
||||
* DNS reverse lookup: reverse lookup of IP´s discovered in order to find hostnames
|
||||
* DNS TDL expansion: TLD dictionary brute force enumeration
|
||||
* DNS TLD expansion: TLD dictionary brute force enumeration
|
||||
|
||||
Modules that require an API key:
|
||||
--------------------------------
|
||||
|
||||
@@ -1,17 +1,43 @@
|
||||
#!/usr/bin/env python3
|
||||
# coding=utf-8
|
||||
from theHarvester.discovery import linkedinsearch
|
||||
from theHarvester.discovery.constants import splitter
|
||||
import pytest
|
||||
import os
|
||||
import re
|
||||
|
||||
|
||||
class TestGetLinks(object):
|
||||
|
||||
def test_splitter(self):
|
||||
results = [
|
||||
'https://www.linkedin.com/in/don-draper-b1045618',
|
||||
'https://www.linkedin.com/in/don-draper-b59210a',
|
||||
'https://www.linkedin.com/in/don-draper-b5bb50b3',
|
||||
'https://www.linkedin.com/in/don-draper-b83ba26',
|
||||
'https://www.linkedin.com/in/don-draper-b854a51'
|
||||
]
|
||||
filtered_results = splitter(results)
|
||||
assert len(filtered_results) == 1
|
||||
|
||||
def test_get_links(self):
|
||||
search = linkedinsearch.SearchLinkedin("facebook.com", '100')
|
||||
search.process()
|
||||
links = search.get_links()
|
||||
for link in links:
|
||||
print(link)
|
||||
assert type(links) == list
|
||||
|
||||
def test_links_linkedin(self):
|
||||
dir_path = os.path.dirname(os.path.realpath(__file__))
|
||||
mock_response = open(dir_path + "/test_linkedin_links.txt")
|
||||
mock_response_content = mock_response.read()
|
||||
mock_response.close()
|
||||
reg_links = re.compile(r"url=https:\/\/www\.linkedin.com(.*?)&")
|
||||
temp = reg_links.findall(mock_response_content)
|
||||
resul = []
|
||||
for regex_item in temp:
|
||||
stripped_url = regex_item.replace("url=", "")
|
||||
resul.append("https://www.linkedin.com" + stripped_url)
|
||||
assert set(resul)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
@@ -0,0 +1,5 @@
|
||||
LinkedIn</a></h3><div class="s"><div class="hJND5c" style="margin-bottom:2px;word-wrap:break-word"><cite>https://www.linkedin.<b>com</b>/in/gm-tuhin-ialam-546526b8</cite></div><div class="f slp">Albany, New York Area - Facebook Advertising</div><span class="st">Gm Tuhin.ialam. <b>facebook</b>.<b>com</b> at Facebook Advertising. Albany, New York Area. <br>
|
||||
Marketing and Advertising. Facebook Advertising. 0 connections ...</span><br></div></div><div class="g"><h3 class="r"><a href="/url?url=https://in.linkedin.com/in/nikulact&rct=j&frm=1&q=&esrc=s&sa=U&ved=0ahUKEwjRpfC-9b_kAhVNnJ4KHcs9C2MQFggcMAM&usg=AOvVaw1Bqufzx2E449oJST9BI5cT">NIKUL www.<b>facebook</b>.<b>com</b>/nikulact - Modeling - Self Modeling ...</a></h3><div class="s"><div class="hJND5c" style="margin-bottom:2px;word-wrap:break-word"><cite>https://in.linkedin.<b>com</b>/in/nikulact</cite><div class="Pj9hGd"><div style="display:inline" onclick="google.sham(this);" aria-expanded="false" aria-haspopup="true" tabindex="0" data-ved="0ahUKEwjRpfC-9b_kAhVNnJ4KHcs9C2MQ7B0IHTAD"><span class="CiacGf"></span></div><div style="display:none" class="am-dropdown-menu" role="menu" tabindex="-1"><ul><li class="mUpfKd"><a class="imx0m" href="/search?hl=en&q=related:https://in.linkedin.com/in/nikulact&tbo=1&sa=X&ved=0ahUKEwjRpfC-9b_kAhVNnJ4KHcs9C2MQHwgfMAM">Similar</a></li></ul></div></div></div><div class="f slp">Ahmedabad Area, India - Self Modeling</div><span class="st">View NIKUL www.<b>facebook</b>.<b>com</b>/nikulact's profile on LinkedIn, the world's largest <br>
|
||||
professional community. NIKUL has 1 job listed on their profile. See the ...</span><br></div></div><div class="g"><h3 class="r"><a href="/url?url=https://www.linkedin.com/in/victor-scott-9a967343&rct=j&frm=1&q=&esrc=s&sa=U&ved=0ahUKEwjRpfC-9b_kAhVNnJ4KHcs9C2MQFggiMAQ&usg=AOvVaw2unq4BLAYGCfUquVZB3R4M">Victor Scott - Metal Band <b>facebook</b>.<b>com</b>/alchemyoftime - Alchemy of ...</a></h3><div class="s"><div class="hJND5c" style="margin-bottom:2px;word-wrap:break-word"><cite>https://www.linkedin.<b>com</b>/in/victor-scott-9a967343</cite></div><div class="f slp">Albany, New York Area - Alchemy of Time</div><span class="st">Victor Scott. Metal Band <b>facebook</b>.<b>com</b>/alchemyoftime at Alchemy of Time. Albany<br>
|
||||
, New York Area. Music. Alchemy of Time. 1 connection ...</span><br></div></div><div class="g"><h3 class="r"><a href="/url?url=https://www.linkedin.com/in/elkhorbat-lkhorbat-6028b33a&rct=j&frm=1&q=&esrc=s&sa=U&ved=0ahUKEwjRpfC-9b_kAhVNnJ4KHcs9C2MQFgglMAU&usg=AOvVaw0CgoQP7h8_4jJ1WkOSC6TB">elkhorbat lkhorbat - http://www.<b>facebook</b>.<b>com</b>/pages/elkhorbat ...</a></h3><div class="s"><div class="hJND5c" style="margin-bottom:2px;word-wrap:break-word"><cite>https://www.linkedin.<b>com</b>/in/elkhorbat-lkhorbat-6028b33a</cite></div><div class="f slp">United States - http://www.facebook.com/pages/elkhorbat/302997479939</div><span class="st">View elkhorbat lkhorbat's profile on LinkedIn, the world's largest professional <br>
|
||||
community. elkhorbat has 1 job listed on their profile. See the complete profile on<br>
|
||||
@@ -38,7 +38,6 @@ def start():
|
||||
hunter, intelx,
|
||||
linkedin, linkedin_links, netcraft, securityTrails, threatcrowd,
|
||||
trello, twitter, vhost, virustotal, yahoo''')
|
||||
parser.add_argument('-x', '--exclude', help='exclude options when using all sources', type=str)
|
||||
|
||||
args = parser.parse_args()
|
||||
try:
|
||||
|
||||
@@ -4,6 +4,31 @@ import random
|
||||
googleUA = 'Mozilla/5.0 (Windows NT 6.2) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/28.0.1464.0 Safari/537.36'
|
||||
|
||||
|
||||
def splitter(links):
|
||||
"""
|
||||
Method that tries to remove duplicates
|
||||
LinkedinLists pulls a lot of profiles with the same name.
|
||||
This method triest to remove duplicates from the list.
|
||||
:param links: list of links to remove duplicates from
|
||||
:return: unique-ish list
|
||||
"""
|
||||
unique_list = []
|
||||
name_check = []
|
||||
for url in links:
|
||||
tail = url.split("/")[-1]
|
||||
if len(tail) == 2 or tail == "zh-cn":
|
||||
tail = url.split("/")[-2]
|
||||
name = tail.split("-")
|
||||
if len(name) > 1:
|
||||
joined_name = name[0] + name[1]
|
||||
else:
|
||||
joined_name = name[0]
|
||||
if joined_name not in name_check:
|
||||
unique_list.append(url)
|
||||
name_check.append(joined_name)
|
||||
return unique_list
|
||||
|
||||
|
||||
def filter(lst):
|
||||
"""
|
||||
Method that filters list
|
||||
|
||||
@@ -36,7 +36,7 @@ class SearchLinkedin:
|
||||
|
||||
def get_links(self):
|
||||
links = myparser.Parser(self.totalresults, self.word)
|
||||
return links.links_linkedin()
|
||||
return splitter(links.links_linkedin())
|
||||
|
||||
def process(self):
|
||||
while self.counter < self.limit:
|
||||
|
||||
+113
-122
@@ -1,16 +1,19 @@
|
||||
# This code is in the public domain, it comes with absolutely no
|
||||
# warranty and you can do absolutely whatever you want with it.
|
||||
# This code is in the public domain, it comes
|
||||
# with absolutely no warranty and you can do
|
||||
# absolutely whatever you want with it.
|
||||
# type: ignore
|
||||
|
||||
__date__ = '17 May 2007'
|
||||
__version__ = '1.7'
|
||||
__date__ = '16 March 2015'
|
||||
__version__ = '1.10'
|
||||
__doc__ = """
|
||||
This is markup.py - a Python module that attempts to
|
||||
make it easier to generate HTML/XML from a Python program
|
||||
in an intuitive, lightweight, customizable and pythonic way.
|
||||
It works with both python 2 and 3.
|
||||
|
||||
The code is in the public domain.
|
||||
|
||||
Version: {0} as of {1}.
|
||||
Version: %s as of %s.
|
||||
|
||||
Documentation and further info is at http://markup.sourceforge.net/
|
||||
|
||||
@@ -18,37 +21,47 @@ Please send bug reports, feature requests, enhancement
|
||||
ideas or questions to nogradi at gmail dot com.
|
||||
|
||||
Installation: drop markup.py somewhere into your Python path.
|
||||
""".format(__version__, __date__)
|
||||
""" % (__version__, __date__)
|
||||
|
||||
|
||||
basestring = str
|
||||
String = str
|
||||
long = int
|
||||
|
||||
# tags which are reserved python keywords will be referred
|
||||
# to by a leading underscore otherwise we end up with a syntax error
|
||||
import keyword
|
||||
|
||||
|
||||
class Element:
|
||||
|
||||
"""This class handles the addition of a new element."""
|
||||
|
||||
def __init__(self, tag, case='lower', parent=None):
|
||||
self.parent = parent
|
||||
|
||||
if case == 'lower':
|
||||
self.tag = tag.lower()
|
||||
else:
|
||||
if case == 'upper':
|
||||
self.tag = tag.upper()
|
||||
elif case == 'lower':
|
||||
self.tag = tag.lower()
|
||||
elif case == 'given':
|
||||
self.tag = tag
|
||||
else:
|
||||
self.tag = tag
|
||||
|
||||
def __call__(self, *args, **kwargs):
|
||||
if len(args) > 1:
|
||||
raise ArgumentError(self.tag)
|
||||
|
||||
# If class_ was defined in parent, it should be added to every element.
|
||||
# if class_ was defined in parent it should be added to every element
|
||||
if self.parent is not None and self.parent.class_ is not None:
|
||||
if 'class_' not in kwargs:
|
||||
kwargs['class_'] = self.parent.class_
|
||||
|
||||
if self.parent is None and len(args) == 1:
|
||||
x = [self.render(self.tag, False, myarg, mydict)
|
||||
for myarg, mydict in _argsdicts(args, kwargs)]
|
||||
x = [self.render(self.tag, False, myarg, mydict) for myarg, mydict in _argsdicts(args, kwargs)]
|
||||
return '\n'.join(x)
|
||||
elif self.parent is None and len(args) == 0:
|
||||
x = [self.render(self.tag, True, myarg, mydict)
|
||||
for myarg, mydict in _argsdicts(args, kwargs)]
|
||||
x = [self.render(self.tag, True, myarg, mydict) for myarg, mydict in _argsdicts(args, kwargs)]
|
||||
return '\n'.join(x)
|
||||
|
||||
if self.tag in self.parent.twotags:
|
||||
@@ -57,8 +70,7 @@ class Element:
|
||||
elif self.tag in self.parent.onetags:
|
||||
if len(args) == 0:
|
||||
for myarg, mydict in _argsdicts(args, kwargs):
|
||||
# Here myarg is always None, because len( args ) = 0.
|
||||
self.render(self.tag, True, myarg, mydict)
|
||||
self.render(self.tag, True, myarg, mydict) # here myarg is always None, because len( args ) = 0
|
||||
else:
|
||||
raise ClosingError(self.tag)
|
||||
elif self.parent.mode == 'strict_html' and self.tag in self.parent.deptags:
|
||||
@@ -70,13 +82,10 @@ class Element:
|
||||
"""Append the actual tags to content."""
|
||||
|
||||
out = "<%s" % tag
|
||||
for key, value in kwargs.items():
|
||||
# When value is None, that means stuff like <... checked>.
|
||||
if value is not None:
|
||||
# Strip this so class_ will mean class, etc.
|
||||
key = key.strip('_')
|
||||
# Special cases, maybe change _ to - overall?
|
||||
if key == 'http_equiv':
|
||||
for key, value in list(kwargs.items()):
|
||||
if value is not None: # when value is None that means stuff like <... checked>
|
||||
key = key.strip('_') # strip this so class_ will mean class, etc.
|
||||
if key == 'http_equiv': # special cases, maybe change _ to - overall?
|
||||
key = 'http-equiv'
|
||||
elif key == 'accept_charset':
|
||||
key = 'accept-charset'
|
||||
@@ -116,11 +125,7 @@ class Element:
|
||||
|
||||
class Page:
|
||||
|
||||
"""This is our main class representing a document. Elements are added as
|
||||
attributes of an instance of this class."""
|
||||
|
||||
def __init__(self, mode='strict_html', case='lower',
|
||||
onetags=None, twotags=None, separator='\n', class_=None):
|
||||
def __init__(self, mode='strict_html', case='lower', onetags=None, twotags=None, separator='\n', class_=None):
|
||||
"""Stuff that effects the whole document.
|
||||
|
||||
mode -- 'strict_html' for HTML 4.01 (default)
|
||||
@@ -130,6 +135,7 @@ class Page:
|
||||
|
||||
case -- 'lower' element names will be printed in lower case (default)
|
||||
'upper' they will be printed in upper case
|
||||
'given' element names will be printed as they are given
|
||||
|
||||
onetags -- list or tuple of valid elements with opening tags only
|
||||
twotags -- list or tuple of valid elements with both opening and closing tags
|
||||
@@ -141,36 +147,16 @@ class Page:
|
||||
|
||||
class_ -- a class that will be added to every element if defined"""
|
||||
|
||||
valid_onetags = [
|
||||
"AREA",
|
||||
"BASE",
|
||||
"BR",
|
||||
"COL",
|
||||
"FRAME",
|
||||
"HR",
|
||||
"IMG",
|
||||
"INPUT",
|
||||
"LINK",
|
||||
"META",
|
||||
"PARAM"]
|
||||
valid_twotags = [
|
||||
"A", "ABBR", "ACRONYM", "ADDRESS", "B", "BDO", "BIG", "BLOCKQUOTE", "BODY", "BUTTON",
|
||||
"CAPTION", "CITE", "CODE", "COLGROUP", "DD", "DEL", "DFN", "DIV", "DL", "DT", "EM", "FIELDSET",
|
||||
"FORM", "FRAMESET", "H1", "H2", "H3", "H4", "H5", "H6", "HEAD", "HTML", "I", "IFRAME", "INS",
|
||||
"KBD", "LABEL", "LEGEND", "LI", "MAP", "NOFRAMES", "NOSCRIPT", "OBJECT", "OL", "OPTGROUP",
|
||||
"OPTION", "P", "PRE", "Q", "SAMP", "SCRIPT", "SELECT", "SMALL", "SPAN", "STRONG", "STYLE",
|
||||
"SUB", "SUP", "TABLE", "TBODY", "TD", "TEXTAREA", "TFOOT", "TH", "THEAD", "TITLE", "TR",
|
||||
"TT", "UL", "VAR"]
|
||||
valid_onetags = ["AREA", "BASE", "BR", "COL", "FRAME", "HR", "IMG", "INPUT", "LINK", "META", "PARAM"]
|
||||
valid_twotags = ["A", "ABBR", "ACRONYM", "ADDRESS", "B", "BDO", "BIG", "BLOCKQUOTE", "BODY", "BUTTON",
|
||||
"CAPTION", "CITE", "CODE", "COLGROUP", "DD", "DEL", "DFN", "DIV", "DL", "DT", "EM", "FIELDSET",
|
||||
"FORM", "FRAMESET", "H1", "H2", "H3", "H4", "H5", "H6", "HEAD", "HTML", "I", "IFRAME", "INS",
|
||||
"KBD", "LABEL", "LEGEND", "LI", "MAP", "NOFRAMES", "NOSCRIPT", "OBJECT", "OL", "OPTGROUP",
|
||||
"OPTION", "P", "PRE", "Q", "SAMP", "SCRIPT", "SELECT", "SMALL", "SPAN", "STRONG", "STYLE",
|
||||
"SUB", "SUP", "TABLE", "TBODY", "TD", "TEXTAREA", "TFOOT", "TH", "THEAD", "TITLE", "TR",
|
||||
"TT", "UL", "VAR"]
|
||||
deprecated_onetags = ["BASEFONT", "ISINDEX"]
|
||||
deprecated_twotags = [
|
||||
"APPLET",
|
||||
"CENTER",
|
||||
"DIR",
|
||||
"FONT",
|
||||
"MENU",
|
||||
"S",
|
||||
"STRIKE",
|
||||
"U"]
|
||||
deprecated_twotags = ["APPLET", "CENTER", "DIR", "FONT", "MENU", "S", "STRIKE", "U"]
|
||||
|
||||
self.header = []
|
||||
self.content = []
|
||||
@@ -178,23 +164,23 @@ class Page:
|
||||
self.case = case
|
||||
self.separator = separator
|
||||
|
||||
# init( ) sets it to True so we know that </body></html> has to be printed at the end.
|
||||
# init( ) sets it to True so we know that </body></html> has to be printed at the end
|
||||
self._full = False
|
||||
self.class_ = class_
|
||||
|
||||
if mode == 'strict_html' or mode == 'html':
|
||||
self.onetags = valid_onetags
|
||||
self.onetags += [str.lower(x) for x in self.onetags]
|
||||
self.onetags += list(map(String.lower, self.onetags))
|
||||
self.twotags = valid_twotags
|
||||
self.twotags += [str.lower(x) for x in self.twotags]
|
||||
self.twotags += list(map(String.lower, self.twotags))
|
||||
self.deptags = deprecated_onetags + deprecated_twotags
|
||||
self.deptags += [str.lower(x) for x in self.deptags]
|
||||
self.deptags += list(map(String.lower, self.deptags))
|
||||
self.mode = 'strict_html'
|
||||
elif mode == 'loose_html':
|
||||
self.onetags = valid_onetags + deprecated_onetags
|
||||
self.onetags += [str.lower(x) for x in self.onetags]
|
||||
self.onetags += list(map(String.lower, self.onetags))
|
||||
self.twotags = valid_twotags + deprecated_twotags
|
||||
self.onetags += [str.lower(x) for x in self.twotags]
|
||||
self.twotags += list(map(String.lower, self.twotags))
|
||||
self.mode = mode
|
||||
elif mode == 'xml':
|
||||
if onetags and twotags:
|
||||
@@ -210,8 +196,16 @@ class Page:
|
||||
raise ModeError(mode)
|
||||
|
||||
def __getattr__(self, attr):
|
||||
|
||||
# tags should start with double underscore
|
||||
if attr.startswith("__") and attr.endswith("__"):
|
||||
raise AttributeError(attr)
|
||||
# tag with single underscore should be a reserved keyword
|
||||
if attr.startswith('_'):
|
||||
attr = attr.lstrip('_')
|
||||
if attr not in keyword.kwlist:
|
||||
raise AttributeError(attr)
|
||||
|
||||
return Element(attr, case=self.case, parent=self)
|
||||
|
||||
def __str__(self):
|
||||
@@ -252,7 +246,7 @@ class Page:
|
||||
self.content.append(text)
|
||||
|
||||
def init(self, lang='en', css=None, metainfo=None, title=None, header=None,
|
||||
footer=None, charset=None, encoding=None, doctype=None, bodyattrs=None, script=None):
|
||||
footer=None, charset=None, encoding=None, doctype=None, bodyattrs=None, script=None, base=None):
|
||||
"""This method is used for complete documents with appropriate
|
||||
doctype, encoding, title, etc information. For an HTML/XML snippet
|
||||
omit this method.
|
||||
@@ -267,11 +261,14 @@ class Page:
|
||||
into meta element(s) as <meta name='name' content='content'>
|
||||
(ignored in xml mode)
|
||||
|
||||
base -- set the <base href="..."> tag in <head>
|
||||
|
||||
bodyattrs --a dictionary in the form { 'key':'value', ... } which will be added
|
||||
as attributes of the <body> element as <body key='value' ... >
|
||||
(ignored in xml mode)
|
||||
|
||||
script -- dictionary containing src:type pairs, <script type='text/type' src=src></script>
|
||||
or a list of [ 'src1', 'src2', ... ] in which case 'javascript' is assumed for all
|
||||
|
||||
title -- the title of the document as a string to be inserted into
|
||||
a title element as <title>my title</title> (ignored in xml mode)
|
||||
@@ -303,10 +300,7 @@ class Page:
|
||||
self.html(lang=lang)
|
||||
self.head()
|
||||
if charset is not None:
|
||||
self.meta(
|
||||
http_equiv='Content-Type',
|
||||
content="text/html; charset=%s" %
|
||||
charset)
|
||||
self.meta(http_equiv='Content-Type', content="text/html; charset=%s" % charset)
|
||||
if metainfo is not None:
|
||||
self.metainfo(metainfo)
|
||||
if css is not None:
|
||||
@@ -315,6 +309,8 @@ class Page:
|
||||
self.title(title)
|
||||
if script is not None:
|
||||
self.scripts(script)
|
||||
if base is not None:
|
||||
self.base(href='%s' % base)
|
||||
self.head.close()
|
||||
if bodyattrs is not None:
|
||||
self.body(**bodyattrs)
|
||||
@@ -334,45 +330,40 @@ class Page:
|
||||
self.header.append(doctype)
|
||||
|
||||
def css(self, filelist):
|
||||
"""This convenience function is only useful for html. It adds CSS
|
||||
stylesheet(s) to the document via the <link> element."""
|
||||
"""This convenience function is only useful for html.
|
||||
It adds css stylesheet(s) to the document via the <link> element."""
|
||||
|
||||
if isinstance(filelist, str):
|
||||
self.link(
|
||||
href=filelist,
|
||||
rel='stylesheet',
|
||||
type='text/css',
|
||||
media='all')
|
||||
if isinstance(filelist, basestring):
|
||||
self.link(href=filelist, rel='stylesheet', type='text/css', media='all')
|
||||
else:
|
||||
for file in filelist:
|
||||
self.link(
|
||||
href=file,
|
||||
rel='stylesheet',
|
||||
type='text/css',
|
||||
media='all')
|
||||
self.link(href=file, rel='stylesheet', type='text/css', media='all')
|
||||
|
||||
def metainfo(self, mydict):
|
||||
"""This convenience function is only useful for html. It adds meta
|
||||
information via the <meta> element, the argument is a dictionary of
|
||||
the form { 'name':'content' }."""
|
||||
"""This convenience function is only useful for html.
|
||||
It adds meta information via the <meta> element, the argument is
|
||||
a dictionary of the form { 'name':'content' }."""
|
||||
|
||||
if isinstance(mydict, dict):
|
||||
for name, content in mydict.items():
|
||||
for name, content in list(mydict.items()):
|
||||
self.meta(name=name, content=content)
|
||||
else:
|
||||
raise TypeError(
|
||||
"Metainfo should be called with a dictionary argument of name:content pairs.")
|
||||
raise TypeError("Metainfo should be called with a dictionary argument of name:content pairs.")
|
||||
|
||||
def scripts(self, mydict):
|
||||
"""Only useful in html, mydict is dictionary of src:type pairs will
|
||||
be rendered as <script type='text/type' src=src></script>"""
|
||||
"""Only useful in html, mydict is dictionary of src:type pairs or a list
|
||||
of script sources [ 'src1', 'src2', ... ] in which case 'javascript' is assumed for type.
|
||||
Will be rendered as <script type='text/type' src=src></script>"""
|
||||
|
||||
if isinstance(mydict, dict):
|
||||
for src, type in mydict.items():
|
||||
for src, type in list(mydict.items()):
|
||||
self.script('', src=src, type='text/%s' % type)
|
||||
else:
|
||||
raise TypeError(
|
||||
"Script should be given a dictionary of src:type pairs.")
|
||||
try:
|
||||
for src in mydict:
|
||||
self.script('', src=src, type='text/javascript')
|
||||
except Exception:
|
||||
raise TypeError("Script should be given a dictionary of src:type pairs or a list of javascript src's.")
|
||||
|
||||
|
||||
class _OneLiner:
|
||||
@@ -385,18 +376,26 @@ class _OneLiner:
|
||||
self.case = case
|
||||
|
||||
def __getattr__(self, attr):
|
||||
|
||||
# tags should start with double underscore
|
||||
if attr.startswith("__") and attr.endswith("__"):
|
||||
raise AttributeError(attr)
|
||||
# tag with single underscore should be a reserved keyword
|
||||
if attr.startswith('_'):
|
||||
attr = attr.lstrip('_')
|
||||
if attr not in keyword.kwlist:
|
||||
raise AttributeError(attr)
|
||||
|
||||
return Element(attr, case=self.case, parent=None)
|
||||
|
||||
|
||||
oneliner = _OneLiner(case='lower')
|
||||
upper_oneliner = _OneLiner(case='upper')
|
||||
given_oneliner = _OneLiner(case='given')
|
||||
|
||||
|
||||
def _argsdicts(args, mydict):
|
||||
"""A utility generator that pads argument list and dictionary values, will
|
||||
only be called with len( args ) = 0, 1."""
|
||||
'''A utility generator that pads argument list and dictionary values, will only be called with len( args ) = 0, 1.'''
|
||||
|
||||
if len(args) == 0:
|
||||
args = None,
|
||||
@@ -407,7 +406,9 @@ def _argsdicts(args, mydict):
|
||||
|
||||
mykeys = list(mydict.keys())
|
||||
myvalues = list(map(_totuple, list(mydict.values())))
|
||||
|
||||
maxlength = max(list(map(len, [args] + myvalues)))
|
||||
|
||||
for i in range(maxlength):
|
||||
thisdict = {}
|
||||
for key, value in zip(mykeys, myvalues):
|
||||
@@ -424,11 +425,11 @@ def _argsdicts(args, mydict):
|
||||
|
||||
|
||||
def _totuple(x):
|
||||
"""Utility stuff to convert string, int, float, None or anything to a usable tuple."""
|
||||
"""Utility stuff to convert string, int, long, float, None or anything to a usable tuple."""
|
||||
|
||||
if isinstance(x, str):
|
||||
if isinstance(x, basestring):
|
||||
out = x,
|
||||
elif isinstance(x, (int, float)):
|
||||
elif isinstance(x, (int, long, float)):
|
||||
out = str(x),
|
||||
elif x is None:
|
||||
out = None,
|
||||
@@ -441,7 +442,7 @@ def _totuple(x):
|
||||
def escape(text, newline=False):
|
||||
"""Escape special html characters."""
|
||||
|
||||
if isinstance(text, str):
|
||||
if isinstance(text, basestring):
|
||||
if '&' in text:
|
||||
text = text.replace('&', '&')
|
||||
if '>' in text:
|
||||
@@ -465,7 +466,7 @@ _escape = escape
|
||||
def unescape(text):
|
||||
"""Inverse of escape."""
|
||||
|
||||
if isinstance(text, str):
|
||||
if isinstance(text, basestring):
|
||||
if '&' in text:
|
||||
text = text.replace('&', '&')
|
||||
if '>' in text:
|
||||
@@ -479,19 +480,17 @@ def unescape(text):
|
||||
|
||||
|
||||
class Dummy:
|
||||
|
||||
"""A dummy class for attaching attributes."""
|
||||
pass
|
||||
|
||||
|
||||
doctype = Dummy()
|
||||
doctype.frameset = "<!DOCTYPE HTML PUBLIC '-//W3C//DTD HTML 4.01 Frameset//EN' 'http://www.w3.org/TR/html4/frameset.dtd'>"
|
||||
doctype.strict = "<!DOCTYPE HTML PUBLIC '-//W3C//DTD HTML 4.01//EN' 'http://www.w3.org/TR/html4/strict.dtd'>"
|
||||
doctype.loose = "<!DOCTYPE HTML PUBLIC '-//W3C//DTD HTML 4.01 Transitional//EN' 'http://www.w3.org/TR/html4/loose.dtd'>"
|
||||
doctype.frameset = """<!DOCTYPE HTML PUBLIC "-//W3C//DTD HTML 4.01 Frameset//EN" "http://www.w3.org/TR/html4/frameset.dtd">"""
|
||||
doctype.strict = """<!DOCTYPE HTML PUBLIC "-//W3C//DTD HTML 4.01//EN" "http://www.w3.org/TR/html4/strict.dtd">"""
|
||||
doctype.loose = """<!DOCTYPE HTML PUBLIC "-//W3C//DTD HTML 4.01 Transitional//EN" "http://www.w3.org/TR/html4/loose.dtd">"""
|
||||
|
||||
|
||||
class Russell:
|
||||
|
||||
"""A dummy class that contains anything."""
|
||||
|
||||
def __contains__(self, item):
|
||||
@@ -499,7 +498,6 @@ class Russell:
|
||||
|
||||
|
||||
class MarkupError(Exception):
|
||||
|
||||
"""All our exceptions subclass this."""
|
||||
|
||||
def __str__(self):
|
||||
@@ -507,48 +505,41 @@ class MarkupError(Exception):
|
||||
|
||||
|
||||
class ClosingError(MarkupError):
|
||||
|
||||
def __init__(self, tag):
|
||||
self.message = "The element '{}' does not accept non-keyword arguments (has no closing tag)".format(tag)
|
||||
self.message = "The element '%s' does not accept non-keyword arguments (has no closing tag)." % tag
|
||||
|
||||
|
||||
class OpeningError(MarkupError):
|
||||
|
||||
def __init__(self, tag):
|
||||
self.message = "The element '{}' can not be opened.".format(tag)
|
||||
self.message = "The element '%s' can not be opened." % tag
|
||||
|
||||
|
||||
class ArgumentError(MarkupError):
|
||||
|
||||
def __init__(self, tag):
|
||||
self.message = "The element '{}' was called with more than one non-keyword argument.".format(tag)
|
||||
self.message = "The element '%s' was called with more than one non-keyword argument." % tag
|
||||
|
||||
|
||||
class InvalidElementError(MarkupError):
|
||||
|
||||
def __init__(self, tag, mode):
|
||||
self.message = "The element '{0}' is not valid for your mode '{1}'.".format(
|
||||
tag,
|
||||
mode)
|
||||
self.message = "The element '%s' is not valid for your mode '%s'." % (tag, mode)
|
||||
|
||||
|
||||
class DeprecationError(MarkupError):
|
||||
|
||||
def __init__(self, tag):
|
||||
self.message = "The element '{0}' is deprecated, instantiate markup.page with mode='loose_html' to allow it.".format(tag)
|
||||
self.message = "The element '%s' is deprecated, instantiate markup.page with mode='loose_html' to allow it." % tag
|
||||
|
||||
|
||||
class ModeError(MarkupError):
|
||||
|
||||
def __init__(self, mode):
|
||||
self.message = "Mode '{}' is invalid, possible values: strict_html, loose_html, xml.".format(mode)
|
||||
self.message = "Mode '%s' is invalid, possible values: strict_html, html (alias for strict_html), loose_html, xml." % mode
|
||||
|
||||
|
||||
class CustomizationError(MarkupError):
|
||||
|
||||
def __init__(self):
|
||||
self.message = "If you customize the allowed elements, you must define both types 'onetags' and 'twotags'."
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
print(__doc__)
|
||||
import sys
|
||||
|
||||
sys.stdout.write(__doc__)
|
||||
|
||||
@@ -4,7 +4,7 @@ class Parser:
|
||||
self.emails = set()
|
||||
self.hosts = set()
|
||||
|
||||
def parse_dictionaries(self, results):
|
||||
def parse_dictionaries(self, results: dict) -> tuple:
|
||||
"""
|
||||
Parse method to parse json results
|
||||
:param results: Dictionary containing a list of dictionaries known as selectors
|
||||
|
||||
@@ -81,10 +81,10 @@ class Parser:
|
||||
reg_links = re.compile(r"url=https:\/\/www\.linkedin.com(.*?)&")
|
||||
self.temp = reg_links.findall(self.results)
|
||||
resul = []
|
||||
for x in self.temp:
|
||||
y = x.replace("url=", "")
|
||||
resul.append("https://www.linkedin.com" + y)
|
||||
return set(resul)
|
||||
for regex in self.temp:
|
||||
final_url = regex.replace("url=", "")
|
||||
resul.append("https://www.linkedin.com" + final_url)
|
||||
return resul
|
||||
|
||||
def people_linkedin(self):
|
||||
reg_people = re.compile(r'">[a-zA-Z0-9._ -]* \| LinkedIn')
|
||||
|
||||
Reference in New Issue
Block a user