[auto-b] extract linked profile (app.py only) - ref e51aa43e, 2846792c, and 86c87382

This commit is contained in:
G
2021-02-27 22:34:31 -08:00
committed by GitHub
parent ad847da2ba
commit 35cfeed60d
3 changed files with 1502 additions and 673 deletions
+30 -4
View File
@@ -25,6 +25,7 @@ from tld import get_fld
from functools import wraps
from bs4 import BeautifulSoup
from re import sub as resub
from re import findall
from copy import deepcopy
from contextlib import suppress
from langdetect import detect
@@ -34,6 +35,7 @@ from random import randint
from tempfile import mkdtemp
from termcolor import colored
from os import system
from urllib.parse import unquote, urlparse
if platform == "win32":
system("color")
@@ -205,7 +207,6 @@ def find_username_normal(req):
"User-Agent": "Mozilla/5.0 (X11; Ubuntu; Linux x86_64; rv:84.0) Gecko/20100101 Firefox/84.0",
}
try:
response = get(site["url"].replace("{username}", username), timeout=5, headers=headers, verify=False)
source = response.text
@@ -216,6 +217,13 @@ def find_username_normal(req):
temp_detected = {}
detections_count = 0
def check_url(url):
with suppress(Exception):
result = urlparse(url)
if result.scheme == "http" or result.scheme == "https":
return all([result.scheme, result.netloc])
return False
def merge_dicts(temp_dict):
result = {}
for item in temp_dict:
@@ -240,6 +248,7 @@ def find_username_normal(req):
"language": "unavailable",
"text": "unavailable",
"type": "unavailable",
"extract":"unavailable",
"good":"",
"method":""
}
@@ -289,6 +298,23 @@ def find_username_normal(req):
temp_profile["title"] = BeautifulSoup(source, "html.parser").title.string
temp_profile["title"] = resub("\s\s+", " ", temp_profile["title"])
with suppress(Exception):
temp_matches = []
if "extract" in site:
for item in site["extract"]:
matches = findall(item["regex"],source)
for match in matches:
if item["type"] == "link":
if check_url(unquote(match)):
parsed ="{}:({})".format(item["type"],unquote(match))
if parsed not in temp_matches:
temp_matches.append(parsed)
if len(temp_matches) > 0:
temp_profile["extract"] = ", ".join(temp_matches)
else:
del temp_profile["extract"]
temp_profile["text"] = temp_profile["text"].replace("\n", "").replace("\t", "").replace("\r", "").strip()
temp_profile["title"] = temp_profile["title"].replace("\n", "").replace("\t", "").replace("\r", "").strip()
@@ -380,7 +406,7 @@ def check_user_cli(argv):
item = clean_up_item(item,argv.options)
temp_detected["detected"].append(item)
else:
item = delete_keys(item,["found","rate","status","method","good"])
item = delete_keys(item,["found","rate","status","method","good","extract"])
item = clean_up_item(item,argv.options)
temp_detected["unknown"].append(item)
elif item["method"] == "find":
@@ -389,11 +415,11 @@ def check_user_cli(argv):
item = clean_up_item(item,argv.options)
temp_detected["detected"].append(item)
elif item["method"] == "get":
item = delete_keys(item,["found","rate","status","method","good"])
item = delete_keys(item,["found","rate","status","method","good", "extract"])
item = clean_up_item(item,argv.options)
temp_detected["unknown"].append(item)
else:
item = delete_keys(item,["found","rate","status","method","good","text","title","language","rate"])
item = delete_keys(item,["found","rate","status","method","good","text","title","language","rate", "extract"])
item = clean_up_item(item,argv.options)
temp_detected["failed"].append(item)
+1469 -666
View File
File diff suppressed because it is too large Load Diff
+3 -3
View File
@@ -1,4 +1,4 @@
{"version":"2021.V.2.27",
{"version":"2021.V.2.28",
"build":"pass",
"test":"pass",
"grid_test":"pass",
@@ -8,7 +8,7 @@
"linux":"pass",
"windows":"pass",
"docker":"pass",
"full_scan":"15 workers < 20secs",
"full_scan":"15 workers < 23secs",
"max_retries":"3",
"awaiting_verification":"22",
"auto_testing":"35c73e1e-f50c-4845-a512-bd6938d8419c"}
"auto_testing":"4779f35e-c96f-4691-a56d-c119901082ee"}