From ba5013fb151e63bcfaec213f7db059883fde8370 Mon Sep 17 00:00:00 2001 From: Raynald Date: Fri, 8 Dec 2023 15:08:18 +0100 Subject: [PATCH 01/12] Remove recursive loop in FFUF & add comments --- web/reNgine/tasks.py | 51 +++++++++++++++++++++++++++++++++----------- 1 file changed, 38 insertions(+), 13 deletions(-) diff --git a/web/reNgine/tasks.py b/web/reNgine/tasks.py index bffec578..aeb13063 100644 --- a/web/reNgine/tasks.py +++ b/web/reNgine/tasks.py @@ -1661,9 +1661,15 @@ def dir_file_fuzz(self, ctx={}, description=None): history_file=self.history_file, scan_id=self.scan_id, activity_id=self.activity_id): + + # Empty line, continue to the next record if not isinstance(line, dict): continue + + # Append line to results results.append(line) + + # Retrieve FFUF output name = line['input'].get('FUZZ') length = line['length'] status = line['status'] @@ -1672,38 +1678,57 @@ def dir_file_fuzz(self, ctx={}, description=None): lines = line['lines'] content_type = line['content-type'] duration = line['duration'] + + # If name empty log error and continue if not name: logger.error(f'FUZZ not found for "{url}"') continue + + # Get or create endpoint from URL endpoint, created = save_endpoint(url, crawl=False, ctx=ctx) + + # Continue to next line if endpoint returned is None + if endpoint == None: + continue + + # Save endpoint data from FFUF output endpoint.http_status = status endpoint.content_length = length endpoint.response_time = duration / 1000000000 - endpoint.save() - if created: - urls.append(endpoint.http_url) - endpoint.status = status endpoint.content_type = content_type endpoint.content_length = length + endpoint.save() + + # Save directory file output from FFUF output dfile, created = DirectoryFile.objects.get_or_create( name=name, length=length, words=words, lines=lines, content_type=content_type, - url=url) - dfile.http_status = status - dfile.save() - # if created: - # logger.warning(f'Found new directory or file {url}') - dirscan.directory_files.add(dfile) - dirscan.save() + url=url, + http_status=status) + # Log newly created file or directory if debug activated + if created and DEBUG: + logger.warning(f'Found new directory or file {url}') + + # Add file to current dirscan + dirscan.directory_files.add(dfile) + + # Add subscan relation to dirscan if exists if self.subscan: dirscan.dir_subscan_ids.add(self.subscan) - subdomain_name = get_subdomain_from_url(endpoint.http_url) - subdomain = Subdomain.objects.get(name=subdomain_name, scan_history=self.scan) + # Save dirscan datas + dirscan.save() + + # Get subdomain and add dirscan + if ctx['subdomain_id'] > 0: + subdomain = Subdomain.objects.get(id=ctx['subdomain_id']) + else: + subdomain_name = get_subdomain_from_url(endpoint.http_url) + subdomain = Subdomain.objects.get(name=subdomain_name, scan_history=self.scan) subdomain.directories.add(dirscan) subdomain.save() From dca6e580f15a5acfd0221307fedff66ec7a90f1d Mon Sep 17 00:00:00 2001 From: Raynald Date: Fri, 8 Dec 2023 17:38:15 +0100 Subject: [PATCH 02/12] Use URL instead of FUZZ to keep full path in name --- web/reNgine/tasks.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/web/reNgine/tasks.py b/web/reNgine/tasks.py index aeb13063..06bdbb4c 100644 --- a/web/reNgine/tasks.py +++ b/web/reNgine/tasks.py @@ -10,6 +10,7 @@ import xmltodict import yaml import tldextract import concurrent.futures +import base64 from datetime import datetime from urllib.parse import urlparse @@ -1670,11 +1671,13 @@ def dir_file_fuzz(self, ctx={}, description=None): results.append(line) # Retrieve FFUF output - name = line['input'].get('FUZZ') + url = line['url'] + url_fuzzed = urlparse(url) + # convert to base64 (need byte string encode & decode) + name = base64.b64encode(url_fuzzed.path.encode()).decode() length = line['length'] status = line['status'] words = line['words'] - url = line['url'] lines = line['lines'] content_type = line['content-type'] duration = line['duration'] From fa7eae1920bbf6f1194b6d62bedc2532a9ff5b61 Mon Sep 17 00:00:00 2001 From: Raynald Date: Sat, 9 Dec 2023 01:19:04 +0100 Subject: [PATCH 03/12] Extract full path from ffuf output --- web/reNgine/common_func.py | 13 +++++++++++++ web/reNgine/tasks.py | 5 ++--- 2 files changed, 15 insertions(+), 3 deletions(-) diff --git a/web/reNgine/common_func.py b/web/reNgine/common_func.py index cb0f1ad1..e8e83d0b 100644 --- a/web/reNgine/common_func.py +++ b/web/reNgine/common_func.py @@ -458,6 +458,19 @@ def sanitize_url(http_url): url = url._replace(netloc=url.netloc.replace(':443', '')) return url.geturl().rstrip('/') +def extract_path_from_url(url): + parsed_url = urlparse(url) + + # Reconstruct the URL without scheme and netloc + reconstructed_url = parsed_url.path + if parsed_url.params: + reconstructed_url += ';' + parsed_url.params + if parsed_url.query: + reconstructed_url += '?' + parsed_url.query + if parsed_url.fragment: + reconstructed_url += '#' + parsed_url.fragment + + return reconstructed_url #-------# # Utils # diff --git a/web/reNgine/tasks.py b/web/reNgine/tasks.py index 06bdbb4c..393cac8d 100644 --- a/web/reNgine/tasks.py +++ b/web/reNgine/tasks.py @@ -1672,9 +1672,8 @@ def dir_file_fuzz(self, ctx={}, description=None): # Retrieve FFUF output url = line['url'] - url_fuzzed = urlparse(url) - # convert to base64 (need byte string encode & decode) - name = base64.b64encode(url_fuzzed.path.encode()).decode() + # Extract path and convert to base64 (need byte string encode & decode) + name = base64.b64encode(extract_path_from_url(url).encode()).decode() length = line['length'] status = line['status'] words = line['words'] From 679be48ba7a9c4d74caec3b44a01cd589738476e Mon Sep 17 00:00:00 2001 From: Raynald Date: Sat, 9 Dec 2023 01:30:48 +0100 Subject: [PATCH 04/12] Remove leading slah in ffuf path extract --- web/reNgine/common_func.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/web/reNgine/common_func.py b/web/reNgine/common_func.py index e8e83d0b..2c7bdbd8 100644 --- a/web/reNgine/common_func.py +++ b/web/reNgine/common_func.py @@ -463,6 +463,10 @@ def extract_path_from_url(url): # Reconstruct the URL without scheme and netloc reconstructed_url = parsed_url.path + + if reconstructed_url.startswith('/'): + reconstructed_url = reconstructed_url[1:] # Remove the first slash + if parsed_url.params: reconstructed_url += ';' + parsed_url.params if parsed_url.query: From 6680fcef9a31db8595dda7963bb743a1d6b5f74f Mon Sep 17 00:00:00 2001 From: Raynald Date: Fri, 8 Dec 2023 15:08:18 +0100 Subject: [PATCH 05/12] Remove recursive loop in FFUF & add comments --- web/reNgine/tasks.py | 51 +++++++++++++++++++++++++++++++++----------- 1 file changed, 38 insertions(+), 13 deletions(-) diff --git a/web/reNgine/tasks.py b/web/reNgine/tasks.py index 613293f8..e43423f9 100644 --- a/web/reNgine/tasks.py +++ b/web/reNgine/tasks.py @@ -1661,9 +1661,15 @@ def dir_file_fuzz(self, ctx={}, description=None): history_file=self.history_file, scan_id=self.scan_id, activity_id=self.activity_id): + + # Empty line, continue to the next record if not isinstance(line, dict): continue + + # Append line to results results.append(line) + + # Retrieve FFUF output name = line['input'].get('FUZZ') length = line['length'] status = line['status'] @@ -1672,38 +1678,57 @@ def dir_file_fuzz(self, ctx={}, description=None): lines = line['lines'] content_type = line['content-type'] duration = line['duration'] + + # If name empty log error and continue if not name: logger.error(f'FUZZ not found for "{url}"') continue + + # Get or create endpoint from URL endpoint, created = save_endpoint(url, crawl=False, ctx=ctx) + + # Continue to next line if endpoint returned is None + if endpoint == None: + continue + + # Save endpoint data from FFUF output endpoint.http_status = status endpoint.content_length = length endpoint.response_time = duration / 1000000000 - endpoint.save() - if created: - urls.append(endpoint.http_url) - endpoint.status = status endpoint.content_type = content_type endpoint.content_length = length + endpoint.save() + + # Save directory file output from FFUF output dfile, created = DirectoryFile.objects.get_or_create( name=name, length=length, words=words, lines=lines, content_type=content_type, - url=url) - dfile.http_status = status - dfile.save() - # if created: - # logger.warning(f'Found new directory or file {url}') - dirscan.directory_files.add(dfile) - dirscan.save() + url=url, + http_status=status) + # Log newly created file or directory if debug activated + if created and DEBUG: + logger.warning(f'Found new directory or file {url}') + + # Add file to current dirscan + dirscan.directory_files.add(dfile) + + # Add subscan relation to dirscan if exists if self.subscan: dirscan.dir_subscan_ids.add(self.subscan) - subdomain_name = get_subdomain_from_url(endpoint.http_url) - subdomain = Subdomain.objects.get(name=subdomain_name, scan_history=self.scan) + # Save dirscan datas + dirscan.save() + + # Get subdomain and add dirscan + if ctx['subdomain_id'] > 0: + subdomain = Subdomain.objects.get(id=ctx['subdomain_id']) + else: + subdomain_name = get_subdomain_from_url(endpoint.http_url) + subdomain = Subdomain.objects.get(name=subdomain_name, scan_history=self.scan) subdomain.directories.add(dirscan) subdomain.save() From 0dd7e7825e8ea19e011c4f0471b21fe69414764c Mon Sep 17 00:00:00 2001 From: Raynald Date: Fri, 8 Dec 2023 17:38:15 +0100 Subject: [PATCH 06/12] Use URL instead of FUZZ to keep full path in name --- web/reNgine/tasks.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/web/reNgine/tasks.py b/web/reNgine/tasks.py index e43423f9..0a0b4c0d 100644 --- a/web/reNgine/tasks.py +++ b/web/reNgine/tasks.py @@ -10,6 +10,7 @@ import xmltodict import yaml import tldextract import concurrent.futures +import base64 from datetime import datetime from urllib.parse import urlparse @@ -1670,11 +1671,13 @@ def dir_file_fuzz(self, ctx={}, description=None): results.append(line) # Retrieve FFUF output - name = line['input'].get('FUZZ') + url = line['url'] + url_fuzzed = urlparse(url) + # convert to base64 (need byte string encode & decode) + name = base64.b64encode(url_fuzzed.path.encode()).decode() length = line['length'] status = line['status'] words = line['words'] - url = line['url'] lines = line['lines'] content_type = line['content-type'] duration = line['duration'] From 3fa96b1383d7ffd43d8097d91a359170bb3c4326 Mon Sep 17 00:00:00 2001 From: Raynald Date: Sat, 9 Dec 2023 01:19:04 +0100 Subject: [PATCH 07/12] Extract full path from ffuf output --- web/reNgine/common_func.py | 13 +++++++++++++ web/reNgine/tasks.py | 5 ++--- 2 files changed, 15 insertions(+), 3 deletions(-) diff --git a/web/reNgine/common_func.py b/web/reNgine/common_func.py index cb0f1ad1..e8e83d0b 100644 --- a/web/reNgine/common_func.py +++ b/web/reNgine/common_func.py @@ -458,6 +458,19 @@ def sanitize_url(http_url): url = url._replace(netloc=url.netloc.replace(':443', '')) return url.geturl().rstrip('/') +def extract_path_from_url(url): + parsed_url = urlparse(url) + + # Reconstruct the URL without scheme and netloc + reconstructed_url = parsed_url.path + if parsed_url.params: + reconstructed_url += ';' + parsed_url.params + if parsed_url.query: + reconstructed_url += '?' + parsed_url.query + if parsed_url.fragment: + reconstructed_url += '#' + parsed_url.fragment + + return reconstructed_url #-------# # Utils # diff --git a/web/reNgine/tasks.py b/web/reNgine/tasks.py index 0a0b4c0d..44df30bd 100644 --- a/web/reNgine/tasks.py +++ b/web/reNgine/tasks.py @@ -1672,9 +1672,8 @@ def dir_file_fuzz(self, ctx={}, description=None): # Retrieve FFUF output url = line['url'] - url_fuzzed = urlparse(url) - # convert to base64 (need byte string encode & decode) - name = base64.b64encode(url_fuzzed.path.encode()).decode() + # Extract path and convert to base64 (need byte string encode & decode) + name = base64.b64encode(extract_path_from_url(url).encode()).decode() length = line['length'] status = line['status'] words = line['words'] From b04d5cfb738a50020c70b9c37b52898d969011d1 Mon Sep 17 00:00:00 2001 From: Raynald Date: Sat, 9 Dec 2023 01:30:48 +0100 Subject: [PATCH 08/12] Remove leading slah in ffuf path extract --- web/reNgine/common_func.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/web/reNgine/common_func.py b/web/reNgine/common_func.py index e8e83d0b..2c7bdbd8 100644 --- a/web/reNgine/common_func.py +++ b/web/reNgine/common_func.py @@ -463,6 +463,10 @@ def extract_path_from_url(url): # Reconstruct the URL without scheme and netloc reconstructed_url = parsed_url.path + + if reconstructed_url.startswith('/'): + reconstructed_url = reconstructed_url[1:] # Remove the first slash + if parsed_url.params: reconstructed_url += ';' + parsed_url.params if parsed_url.query: From abe6def19e46b264606ed8aac63a1e1157c6a066 Mon Sep 17 00:00:00 2001 From: Raynald Date: Fri, 8 Dec 2023 15:08:18 +0100 Subject: [PATCH 09/12] Remove recursive loop in FFUF & add comments --- web/reNgine/tasks.py | 51 +++++++++++++++++++++++++++++++++----------- 1 file changed, 38 insertions(+), 13 deletions(-) diff --git a/web/reNgine/tasks.py b/web/reNgine/tasks.py index 52d969e9..64845d90 100644 --- a/web/reNgine/tasks.py +++ b/web/reNgine/tasks.py @@ -1662,9 +1662,15 @@ def dir_file_fuzz(self, ctx={}, description=None): history_file=self.history_file, scan_id=self.scan_id, activity_id=self.activity_id): + + # Empty line, continue to the next record if not isinstance(line, dict): continue + + # Append line to results results.append(line) + + # Retrieve FFUF output name = line['input'].get('FUZZ') length = line['length'] status = line['status'] @@ -1673,38 +1679,57 @@ def dir_file_fuzz(self, ctx={}, description=None): lines = line['lines'] content_type = line['content-type'] duration = line['duration'] + + # If name empty log error and continue if not name: logger.error(f'FUZZ not found for "{url}"') continue + + # Get or create endpoint from URL endpoint, created = save_endpoint(url, crawl=False, ctx=ctx) + + # Continue to next line if endpoint returned is None + if endpoint == None: + continue + + # Save endpoint data from FFUF output endpoint.http_status = status endpoint.content_length = length endpoint.response_time = duration / 1000000000 - endpoint.save() - if created: - urls.append(endpoint.http_url) - endpoint.status = status endpoint.content_type = content_type endpoint.content_length = length + endpoint.save() + + # Save directory file output from FFUF output dfile, created = DirectoryFile.objects.get_or_create( name=name, length=length, words=words, lines=lines, content_type=content_type, - url=url) - dfile.http_status = status - dfile.save() - # if created: - # logger.warning(f'Found new directory or file {url}') - dirscan.directory_files.add(dfile) - dirscan.save() + url=url, + http_status=status) + # Log newly created file or directory if debug activated + if created and DEBUG: + logger.warning(f'Found new directory or file {url}') + + # Add file to current dirscan + dirscan.directory_files.add(dfile) + + # Add subscan relation to dirscan if exists if self.subscan: dirscan.dir_subscan_ids.add(self.subscan) - subdomain_name = get_subdomain_from_url(endpoint.http_url) - subdomain = Subdomain.objects.get(name=subdomain_name, scan_history=self.scan) + # Save dirscan datas + dirscan.save() + + # Get subdomain and add dirscan + if ctx['subdomain_id'] > 0: + subdomain = Subdomain.objects.get(id=ctx['subdomain_id']) + else: + subdomain_name = get_subdomain_from_url(endpoint.http_url) + subdomain = Subdomain.objects.get(name=subdomain_name, scan_history=self.scan) subdomain.directories.add(dirscan) subdomain.save() From 8c1f2b4efbadfd195310c5c41c0a1225ed5e3aa0 Mon Sep 17 00:00:00 2001 From: Raynald Date: Fri, 8 Dec 2023 17:38:15 +0100 Subject: [PATCH 10/12] Use URL instead of FUZZ to keep full path in name --- web/reNgine/tasks.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/web/reNgine/tasks.py b/web/reNgine/tasks.py index 64845d90..efd6df8d 100644 --- a/web/reNgine/tasks.py +++ b/web/reNgine/tasks.py @@ -10,6 +10,7 @@ import xmltodict import yaml import tldextract import concurrent.futures +import base64 from datetime import datetime from urllib.parse import urlparse @@ -1671,11 +1672,13 @@ def dir_file_fuzz(self, ctx={}, description=None): results.append(line) # Retrieve FFUF output - name = line['input'].get('FUZZ') + url = line['url'] + url_fuzzed = urlparse(url) + # convert to base64 (need byte string encode & decode) + name = base64.b64encode(url_fuzzed.path.encode()).decode() length = line['length'] status = line['status'] words = line['words'] - url = line['url'] lines = line['lines'] content_type = line['content-type'] duration = line['duration'] From 6c138c7742c809779cdc61f8b3e1f4ba227953c2 Mon Sep 17 00:00:00 2001 From: Raynald Date: Sat, 9 Dec 2023 01:19:04 +0100 Subject: [PATCH 11/12] Extract full path from ffuf output --- web/reNgine/common_func.py | 13 +++++++++++++ web/reNgine/tasks.py | 5 ++--- 2 files changed, 15 insertions(+), 3 deletions(-) diff --git a/web/reNgine/common_func.py b/web/reNgine/common_func.py index cb0f1ad1..e8e83d0b 100644 --- a/web/reNgine/common_func.py +++ b/web/reNgine/common_func.py @@ -458,6 +458,19 @@ def sanitize_url(http_url): url = url._replace(netloc=url.netloc.replace(':443', '')) return url.geturl().rstrip('/') +def extract_path_from_url(url): + parsed_url = urlparse(url) + + # Reconstruct the URL without scheme and netloc + reconstructed_url = parsed_url.path + if parsed_url.params: + reconstructed_url += ';' + parsed_url.params + if parsed_url.query: + reconstructed_url += '?' + parsed_url.query + if parsed_url.fragment: + reconstructed_url += '#' + parsed_url.fragment + + return reconstructed_url #-------# # Utils # diff --git a/web/reNgine/tasks.py b/web/reNgine/tasks.py index efd6df8d..c074c5f9 100644 --- a/web/reNgine/tasks.py +++ b/web/reNgine/tasks.py @@ -1673,9 +1673,8 @@ def dir_file_fuzz(self, ctx={}, description=None): # Retrieve FFUF output url = line['url'] - url_fuzzed = urlparse(url) - # convert to base64 (need byte string encode & decode) - name = base64.b64encode(url_fuzzed.path.encode()).decode() + # Extract path and convert to base64 (need byte string encode & decode) + name = base64.b64encode(extract_path_from_url(url).encode()).decode() length = line['length'] status = line['status'] words = line['words'] From 5cdf008dda639529e128074e708d80031ae51669 Mon Sep 17 00:00:00 2001 From: Raynald Date: Sat, 9 Dec 2023 01:30:48 +0100 Subject: [PATCH 12/12] Remove leading slah in ffuf path extract --- web/reNgine/common_func.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/web/reNgine/common_func.py b/web/reNgine/common_func.py index e8e83d0b..2c7bdbd8 100644 --- a/web/reNgine/common_func.py +++ b/web/reNgine/common_func.py @@ -463,6 +463,10 @@ def extract_path_from_url(url): # Reconstruct the URL without scheme and netloc reconstructed_url = parsed_url.path + + if reconstructed_url.startswith('/'): + reconstructed_url = reconstructed_url[1:] # Remove the first slash + if parsed_url.params: reconstructed_url += ';' + parsed_url.params if parsed_url.query: