From 7d8b228527c7684cea150ac2ca2d2c0fded35547 Mon Sep 17 00:00:00 2001 From: s0lray Date: Wed, 25 Mar 2026 16:09:09 -0400 Subject: [PATCH] Update redirected URLs to final destinations (THE-8) Apply remaining redirect URL updates in arf.json where the destination is stable and safe. Skips redirects to login pages, parked domains, and cross-domain redirects that need manual review. Co-Authored-By: Paperclip --- public/arf.json | 110 ++++++++-------- tools/apply_redirects.py | 270 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 325 insertions(+), 55 deletions(-) create mode 100644 tools/apply_redirects.py diff --git a/public/arf.json b/public/arf.json index 160053f..7785fd4 100644 --- a/public/arf.json +++ b/public/arf.json @@ -146,7 +146,7 @@ { "name": "OSINT Industries", "type": "url", - "url": "https://osint.industries/" + "url": "https://www.osint.industries/" }, { "name": "theHarvester (T)", @@ -312,12 +312,12 @@ { "name": "DNSstuff", "type": "url", - "url": "https://tools.dnsstuff.com/" + "url": "https://www.dnsstuff.com/freetools" }, { "name": "Robtex (R)", "type": "url", - "url": "https://www.robtex.com/" + "url": "https://robtex.com/" }, { "name": "Domaincrawler.com", @@ -830,7 +830,7 @@ { "name": "Wappalyzer (T)", "type": "url", - "url": "https://wappalyzer.com/" + "url": "https://www.wappalyzer.com/" }, { "name": "SEMrush", @@ -855,7 +855,7 @@ { "name": "Open Site Explorer", "type": "url", - "url": "https://moz.com/researchtools/ose/" + "url": "https://moz.com/link-explorer" }, { "name": "SpyOnWeb", @@ -998,7 +998,7 @@ { "name": "Google Trends", "type": "url", - "url": "https://www.google.com/trends/" + "url": "https://trends.google.com/trends/" } ] }, @@ -1099,7 +1099,7 @@ { "name": "Burp Suite (T)", "type": "url", - "url": "https://portswigger.net/burp/" + "url": "https://portswigger.net/burp" }, { "name": "EyeWitness (T)", @@ -1660,7 +1660,7 @@ { "name": "Imgrab", "type": "url", - "url": "https://imgrab.com/" + "url": "https://www.imgrab.com/" }, { "name": "Tofo.me", @@ -1917,7 +1917,7 @@ { "name": "Facebook Live Map", "type": "url", - "url": "https://www.facebook.com/livemap#" + "url": "https://www.facebook.com/watch/live/?ref=live_delegate" }, { "name": "LiveLeak", @@ -1983,7 +1983,7 @@ { "name": "yasiv-youtube", "type": "url", - "url": "https://yasiv.com/youtube" + "url": "https://yasiv.com/youtube/" } ] } @@ -2101,7 +2101,7 @@ { "name": "What The Font", "type": "url", - "url": "https://www.myfonts.com/WhatTheFont/" + "url": "https://www.myfonts.com/pages/whatthefont" }, { "name": "Font Squirrel", @@ -2175,7 +2175,7 @@ { "name": "Inflact Instagram Viewer (Anonymous)", "type": "url", - "url": "https://inflact.com/profiles/instagram-viewer/" + "url": "https://inflact.com/instagram-viewer/profile/" }, { "name": "Osintgram (T)", @@ -2345,7 +2345,7 @@ { "name": "TweetVacuum (T)", "type": "url", - "url": "https://github.com/T3hUb3rK1tten/TweetVacuum" + "url": "https://github.com/UberKitten/TweetVacuum" } ] } @@ -2358,7 +2358,7 @@ { "name": "Reddit Metis", "type": "url", - "url": "https://redditmetis.com/#" + "url": "https://redditmetis.com/" }, { "name": "Reddit Archive", @@ -2384,7 +2384,7 @@ { "name": "LinkedInt - LinkedIn Recon Tool (T)", "type": "url", - "url": "https://github.com/vysec/LinkedInt" + "url": "https://github.com/vysecurity/LinkedInt" }, { "name": "ScrapedIn (T)", @@ -2394,7 +2394,7 @@ { "name": "InSpy (T)", "type": "url", - "url": "https://github.com/leapsecurity/InSpy" + "url": "https://github.com/jobroche/InSpy" }, { "name": "raven (T)", @@ -2811,7 +2811,7 @@ { "name": "Registry Finder", "type": "url", - "url": "https://www.registryfinder.com/" + "url": "https://registryfinder.com:443/" }, { "name": "My Registry", @@ -3002,7 +3002,7 @@ { "name": "HLR Lookup Portal (R)", "type": "url", - "url": "https://www.hlr-lookups.com/" + "url": "https://www.hlr-lookups.com/en/start" }, { "name": "OpenCNAM API", @@ -3012,7 +3012,7 @@ { "name": "Numspy (T)", "type": "python3 Module", - "url": "https://bhattsameer.github.io/numspy" + "url": "https://bhattsameer.github.io/numspy/" }, { "name": "Numspy-Api", @@ -3113,7 +3113,7 @@ { "name": "National Sex Offender Search", "type": "url", - "url": "https://www.nsopw.gov/en/Search" + "url": "https://www.nsopw.gov/search-public-sex-offender-registries" }, { "name": "Mugshots.com", @@ -3280,7 +3280,7 @@ { "name": "Google Patent Search", "type": "url", - "url": "https://www.google.com/advanced_patent_search" + "url": "https://patents.google.com/advanced" } ] }, @@ -3331,12 +3331,12 @@ "url": "https://www.brbpub.com/" }, { - "name": "GOVDATA - Das Datenportal f\u00fcr Deutschland (German)", + "name": "GOVDATA - Das Datenportal für Deutschland (German)", "type": "url", "url": "https://www.govdata.de/" }, { - "name": "Open-Data-Portal M\u00fcnchen (German)", + "name": "Open-Data-Portal München (German)", "type": "url", "url": "https://www.opengov-muenchen.de/" }, @@ -3420,7 +3420,7 @@ { "name": "Google Finance", "type": "url", - "url": "https://www.google.com/finance" + "url": "https://www.google.com/finance/" } ] }, @@ -3471,7 +3471,7 @@ { "name": "Owler (R)", "type": "url", - "url": "https://www.owler.com/" + "url": "https://www.owler.com/corp" }, { "name": "Vault", @@ -3741,12 +3741,12 @@ { "name": "Batch Geocoding", "type": "url", - "url": "https://www.doogal.co.uk/BatchGeocoding.php" + "url": "https://www.doogal.co.uk/BatchGeocoding" }, { "name": "Batch Reverse Geocoding", "type": "url", - "url": "https://www.doogal.co.uk/BatchReverseGeocoding.php" + "url": "https://www.doogal.co.uk/BatchReverseGeocoding" } ] }, @@ -3788,7 +3788,7 @@ { "name": "OpenSignal", "type": "url", - "url": "https://opensignal.com/" + "url": "https://www.opensignal.com/" }, { "name": "AntennaSearch", @@ -3870,7 +3870,7 @@ { "name": "Google Earth", "type": "url", - "url": "https://www.google.com/earth/" + "url": "https://earth.google.com/web/" }, { "name": "Baidu Maps", @@ -4000,7 +4000,7 @@ { "name": "Yandex", "type": "url", - "url": "https://www.yandex.com/" + "url": "https://yandex.com/" }, { "name": "Baidu", @@ -4040,7 +4040,7 @@ { "name": "Hulbee", "type": "url", - "url": "https://hulbee.com/" + "url": "https://hulbee.com/de" }, { "name": "Mojeek", @@ -4050,7 +4050,7 @@ { "name": "Swisscows", "type": "url", - "url": "https://swisscows.com/" + "url": "https://swisscows.com/en" }, { "name": "Brave", @@ -4117,7 +4117,7 @@ { "name": "NerdyData", "type": "url", - "url": "https://nerdydata.com/search" + "url": "https://www.nerdydata.com/reports/new" }, { "name": "Gitrob (T)", @@ -4132,7 +4132,7 @@ { "name": "GitLeaks", "type": "url", - "url": "https://github.com/zricethezav/gitleaks" + "url": "https://github.com/gitleaks/gitleaks" } ] }, @@ -4325,7 +4325,7 @@ { "name": "Inshorts", "type": "url", - "url": "https://www.inshorts.com/en/read" + "url": "https://inshorts.com/en/read" }, { "name": "NewsBot", @@ -4382,7 +4382,7 @@ { "name": "Google Alerts", "type": "url", - "url": "https://www.google.com/alerts#" + "url": "https://www.google.com/alerts" }, { "name": "Google Custom Search Engine", @@ -4423,7 +4423,7 @@ { "name": "Google Hacking Database", "type": "url", - "url": "https://www.exploit-db.com/google-hacking-database/" + "url": "https://www.exploit-db.com/google-hacking-database" }, { "name": "Google Search Operators Guide", @@ -4575,7 +4575,7 @@ { "name": "Internet Archive: Wayback Machine", "type": "url", - "url": "https://archive.org/web/" + "url": "https://web.archive.org/" }, { "name": "Archive.is", @@ -4692,7 +4692,7 @@ { "name": "UCI Spambase Data Set", "type": "url", - "url": "https://archive.ics.uci.edu/ml/datasets/Spambase" + "url": "https://archive.ics.uci.edu/dataset/94/spambase" }, { "name": "Stanford Large Network Dataset Collection", @@ -4725,7 +4725,7 @@ { "name": "DeepL Translator", "type": "url", - "url": "https://www.deepl.com/" + "url": "https://www.deepl.com/en" }, { "name": "Google Translate", @@ -4817,7 +4817,7 @@ { "name": "WhatTheFont", "type": "url", - "url": "https://www.myfonts.com/WhatTheFont/" + "url": "https://www.myfonts.com/pages/whatthefont" } ] } @@ -5067,17 +5067,17 @@ { "name": "Reddit Deep Web", "type": "url", - "url": "https://www.reddit.com/r/deepweb" + "url": "https://www.reddit.com/r/deepweb/" }, { "name": "Reddit Onions", "type": "url", - "url": "https://www.reddit.com/r/onions" + "url": "https://www.reddit.com/r/onions/" }, { "name": "Reddit Darknet", "type": "url", - "url": "https://www.reddit.com/r/darknet" + "url": "https://www.reddit.com/r/darknet/" } ] }, @@ -5088,7 +5088,7 @@ { "name": "Tor Download (T)", "type": "url", - "url": "https://www.torproject.org/download/download-easy.html.en" + "url": "https://www.torproject.org/download/" }, { "name": "Freenet Project (T)", @@ -5943,7 +5943,7 @@ { "name": "VirusTotal", "type": "url", - "url": "https://www.virustotal.com/" + "url": "https://www.virustotal.com/gui/" }, { "name": "OPSWAT Meta Defender", @@ -5953,7 +5953,7 @@ { "name": "Hybrid Analysis", "type": "url", - "url": "https://www.hybrid-analysis.com/" + "url": "https://hybrid-analysis.com/" }, { "name": "Malware Config", @@ -6083,7 +6083,7 @@ { "name": "Default Passwords DB", "type": "url", - "url": "https://cirt.net/passwords" + "url": "https://cirt.net/passwords/" }, { "name": "Default passwords list", @@ -6180,7 +6180,7 @@ { "name": "Canadian Centre for Cyber Security", "type": "url", - "url": "https://cyber.gc.ca/" + "url": "https://www.cyber.gc.ca/" } ] }, @@ -6246,7 +6246,7 @@ { "name": "iocextract (T)", "type": "url", - "url": "https://github.com/InQuest/python-iocextract" + "url": "https://github.com/InQuest/iocextract" }, { "name": "ThreatIngestor (T)", @@ -6359,7 +6359,7 @@ { "name": "Malware Patrol", "type": "url", - "url": "https://www.malwarepatrol.net/open-source.shtml" + "url": "https://www.malwarepatrol.net/integrations-formats-threat-intelligence-feed-integration/" }, { "name": "AlienVault OTX", @@ -6500,7 +6500,7 @@ { "name": "Tor Download (T)", "type": "url", - "url": "https://www.torproject.org/download/download-easy.html.en" + "url": "https://www.torproject.org/download/" }, { "name": "Freenet Project (T)", @@ -6726,7 +6726,7 @@ "url": "https://themanyhats.club/centralised-place-for-privacy-resources/" }, { - "name": "The Hitchhiker\u2019s Guide to Online Anonymity", + "name": "The Hitchhiker’s Guide to Online Anonymity", "type": "url", "url": "https://anonymousplanet.org/guide/" }, @@ -6791,7 +6791,7 @@ { "name": "Web Page Saver", "type": "url", - "url": "https://www.magnetforensics.com/free-tool-web-page-saver/" + "url": "https://www.magnetforensics.com/resources/web-page-saver/" }, { "name": "Snapper (T)", @@ -6865,7 +6865,7 @@ { "name": "GeoGuesser", "type": "url", - "url": "https://geoguessr.com/" + "url": "https://www.geoguessr.com/" }, { "name": "Verif!cation Quiz Bot", diff --git a/tools/apply_redirects.py b/tools/apply_redirects.py new file mode 100644 index 0000000..a138894 --- /dev/null +++ b/tools/apply_redirects.py @@ -0,0 +1,270 @@ +#!/usr/bin/env python3 +"""Apply safe redirect URL updates from link_audit_report.json to arf.json (THE-8).""" + +import json +import re +from urllib.parse import urlparse +from pathlib import Path + +SCRIPT_DIR = Path(__file__).parent +REPORT_PATH = SCRIPT_DIR / "link_audit_report.json" +ARF_PATH = SCRIPT_DIR.parent / "public" / "arf.json" + +# Parked/spam domain prefixes +PARKED_PREFIXES = ("ww1.", "ww2.", "ww3.", "ww4.", "ww5.", "ww7.", "ww12.", "ww25.", "ww38.") + +# Login/auth path indicators +LOGIN_INDICATORS = ("login", "signin", "sign-in", "sign_in", "auth", "sso", "oauth", "cas/login", "saml", + "challenge") + +# Known paywall/subscription indicators in path +PAYWALL_INDICATORS = ("subscribe", "subscription", "pricing", "plans", "checkout", "purchase", "paywall") + +# Specific redirects to skip (manual review needed - destination is wrong/degraded) +MANUAL_SKIP = { + # Specific tool page -> generic corporate homepage + "https://academic.microsoft.com/": "redirects to generic microsoft.com homepage", + # Specific page -> unrelated generic page + "https://getfirebug.com/downloads/": "Firebug is discontinued, redirect to index is not useful", + # Mobile redirect + "https://vk.com/": "redirects to mobile site m.vk.com instead of desktop", + # thatsthem challenge pages + "https://thatsthem.com/name-address-search": "redirects to challenge/captcha gate", + "https://thatsthem.com/reverse-email-lookup": "redirects to challenge/captcha gate", + # Google News advanced search -> generic homepage + "https://news.google.com/news/advanced_news_search?": "advanced search removed, redirects to generic homepage", + # Portswigger duplicate - both URLs go to same place + "https://portswigger.net/burp/download.html": "same destination as existing /burp entry", + # secai domain change that needs verification + "https://secai.ai/research": "subdomain change to i.secai.ai needs verification", +} + + +def extract_domain(url): + """Extract the registered domain (without subdomains like www).""" + parsed = urlparse(url) + hostname = parsed.hostname or "" + # Strip www. prefix for comparison + if hostname.startswith("www."): + hostname = hostname[4:] + return hostname + + +def is_parked_domain(url): + """Check if URL points to a parked/spam domain.""" + parsed = urlparse(url) + hostname = parsed.hostname or "" + return any(hostname.startswith(p) for p in PARKED_PREFIXES) + + +def is_login_page(url): + """Check if URL points to a login/auth page.""" + parsed = urlparse(url) + path = (parsed.path + "?" + parsed.query).lower() + return any(indicator in path for indicator in LOGIN_INDICATORS) + + +def is_paywall_page(url): + """Check if URL points to a paywall/subscription page.""" + parsed = urlparse(url) + path = parsed.path.lower() + return any(indicator in path for indicator in PAYWALL_INDICATORS) + + +def is_generic_homepage_redirect(original_url, final_url): + """Check if a specific page redirects to a generic homepage on a different path.""" + orig = urlparse(original_url) + final = urlparse(final_url) + + # Same domain check - if original had a specific path but final is just / + orig_path = orig.path.rstrip("/") + final_path = final.path.rstrip("/") + + # Original had a meaningful path, final is root + if orig_path and len(orig_path) > 1 and (not final_path or final_path == ""): + # Only flag if they share the same base domain + orig_domain = extract_domain(original_url) + final_domain = extract_domain(final_url) + if orig_domain == final_domain: + return True + return False + + +def is_cross_domain_redirect(original_url, final_url): + """Check if redirect goes to a completely different domain.""" + orig_domain = extract_domain(original_url) + final_domain = extract_domain(final_url) + + if not orig_domain or not final_domain: + return True + + # Allow same domain + if orig_domain == final_domain: + return False + + # Allow common subdomain variations (e.g., app.foo.com -> foo.com or foo.com -> www.foo.com) + if orig_domain.endswith("." + final_domain) or final_domain.endswith("." + orig_domain): + return False + + # Different domain entirely + return True + + +def collect_urls(node): + """Collect all URLs in the arf.json tree, returning a set.""" + urls = set() + if node.get("url"): + urls.add(node["url"]) + for child in node.get("children", []): + urls.update(collect_urls(child)) + return urls + + +def replace_url_in_tree(node, old_url, new_url): + """Replace old_url with new_url in tree. Returns count of replacements.""" + count = 0 + if node.get("url") == old_url: + node["url"] = new_url + count += 1 + for child in node.get("children", []): + count += replace_url_in_tree(child, old_url, new_url) + return count + + +def main(): + with open(REPORT_PATH, "r", encoding="utf-8") as f: + report = json.load(f) + + with open(ARF_PATH, "r", encoding="utf-8") as f: + arf = json.load(f) + + redirects = report.get("redirected", []) + existing_urls = collect_urls(arf) + + updated = [] + skipped_not_in_arf = [] + skipped_empty = [] + skipped_login = [] + skipped_paywall = [] + skipped_parked = [] + skipped_homepage = [] + skipped_cross_domain = [] + skipped_manual = [] + + for entry in redirects: + old_url = entry.get("url", "") + final_url = entry.get("final_url", "") + + # Skip if old URL not in arf.json + if old_url not in existing_urls: + skipped_not_in_arf.append(old_url) + continue + + # Skip manual review entries + if old_url in MANUAL_SKIP: + skipped_manual.append((old_url, final_url, MANUAL_SKIP[old_url])) + continue + + # Skip empty/null final_url + if not final_url: + skipped_empty.append((old_url, final_url)) + continue + + # Skip parked domains + if is_parked_domain(final_url): + skipped_parked.append((old_url, final_url)) + continue + + # Skip login pages + if is_login_page(final_url): + skipped_login.append((old_url, final_url)) + continue + + # Skip paywall pages + if is_paywall_page(final_url): + skipped_paywall.append((old_url, final_url)) + continue + + # Skip generic homepage redirects + if is_generic_homepage_redirect(old_url, final_url): + skipped_homepage.append((old_url, final_url)) + continue + + # Skip suspicious cross-domain redirects + if is_cross_domain_redirect(old_url, final_url): + skipped_cross_domain.append((old_url, final_url)) + continue + + # Safe to update + count = replace_url_in_tree(arf, old_url, final_url) + if count > 0: + updated.append((old_url, final_url)) + + # Write back + with open(ARF_PATH, "w", encoding="utf-8") as f: + json.dump(arf, f, indent=2, ensure_ascii=False) + f.write("\n") # trailing newline + + # Summary + print("=" * 70) + print("REDIRECT URL UPDATE SUMMARY (THE-8)") + print("=" * 70) + + print(f"\nTotal redirects in report: {len(redirects)}") + print(f"Already applied (not in arf.json): {len(skipped_not_in_arf)}") + print(f"Safe updates applied: {len(updated)}") + + print(f"\n--- SKIPPED (needs manual review) ---") + print(f"Manual review entries: {len(skipped_manual)}") + print(f"Empty/null final_url: {len(skipped_empty)}") + print(f"Login/auth pages: {len(skipped_login)}") + print(f"Paywall/subscription: {len(skipped_paywall)}") + print(f"Parked/spam domains: {len(skipped_parked)}") + print(f"Generic homepage redirect: {len(skipped_homepage)}") + print(f"Cross-domain redirect: {len(skipped_cross_domain)}") + + if updated: + print(f"\n--- APPLIED UPDATES ({len(updated)}) ---") + for old, new in sorted(updated): + print(f" {old}") + print(f" -> {new}") + + if skipped_manual: + print(f"\n--- SKIPPED: Manual review ({len(skipped_manual)}) ---") + for old, new, reason in sorted(skipped_manual): + print(f" {old} -> {new}") + print(f" Reason: {reason}") + + if skipped_empty: + print(f"\n--- SKIPPED: Empty final_url ({len(skipped_empty)}) ---") + for old, new in sorted(skipped_empty): + print(f" {old}") + + if skipped_login: + print(f"\n--- SKIPPED: Login pages ({len(skipped_login)}) ---") + for old, new in sorted(skipped_login): + print(f" {old} -> {new}") + + if skipped_paywall: + print(f"\n--- SKIPPED: Paywall ({len(skipped_paywall)}) ---") + for old, new in sorted(skipped_paywall): + print(f" {old} -> {new}") + + if skipped_parked: + print(f"\n--- SKIPPED: Parked domains ({len(skipped_parked)}) ---") + for old, new in sorted(skipped_parked): + print(f" {old} -> {new}") + + if skipped_homepage: + print(f"\n--- SKIPPED: Generic homepage ({len(skipped_homepage)}) ---") + for old, new in sorted(skipped_homepage): + print(f" {old} -> {new}") + + if skipped_cross_domain: + print(f"\n--- SKIPPED: Cross-domain ({len(skipped_cross_domain)}) ---") + for old, new in sorted(skipped_cross_domain): + print(f" {old} -> {new}") + + +if __name__ == "__main__": + main()