Update redirected URLs to final destinations (THE-8)

Apply remaining redirect URL updates in arf.json where the destination
is stable and safe. Skips redirects to login pages, parked domains,
and cross-domain redirects that need manual review.

Co-Authored-By: Paperclip <noreply@paperclip.ing>
This commit is contained in:
s0lray
2026-03-25 16:09:09 -04:00
co-authored by Paperclip
parent d399af21ab
commit 7d8b228527
2 changed files with 325 additions and 55 deletions
+55 -55
View File
@@ -146,7 +146,7 @@
{
"name": "OSINT Industries",
"type": "url",
"url": "https://osint.industries/"
"url": "https://www.osint.industries/"
},
{
"name": "theHarvester (T)",
@@ -312,12 +312,12 @@
{
"name": "DNSstuff",
"type": "url",
"url": "https://tools.dnsstuff.com/"
"url": "https://www.dnsstuff.com/freetools"
},
{
"name": "Robtex (R)",
"type": "url",
"url": "https://www.robtex.com/"
"url": "https://robtex.com/"
},
{
"name": "Domaincrawler.com",
@@ -830,7 +830,7 @@
{
"name": "Wappalyzer (T)",
"type": "url",
"url": "https://wappalyzer.com/"
"url": "https://www.wappalyzer.com/"
},
{
"name": "SEMrush",
@@ -855,7 +855,7 @@
{
"name": "Open Site Explorer",
"type": "url",
"url": "https://moz.com/researchtools/ose/"
"url": "https://moz.com/link-explorer"
},
{
"name": "SpyOnWeb",
@@ -998,7 +998,7 @@
{
"name": "Google Trends",
"type": "url",
"url": "https://www.google.com/trends/"
"url": "https://trends.google.com/trends/"
}
]
},
@@ -1099,7 +1099,7 @@
{
"name": "Burp Suite (T)",
"type": "url",
"url": "https://portswigger.net/burp/"
"url": "https://portswigger.net/burp"
},
{
"name": "EyeWitness (T)",
@@ -1660,7 +1660,7 @@
{
"name": "Imgrab",
"type": "url",
"url": "https://imgrab.com/"
"url": "https://www.imgrab.com/"
},
{
"name": "Tofo.me",
@@ -1917,7 +1917,7 @@
{
"name": "Facebook Live Map",
"type": "url",
"url": "https://www.facebook.com/livemap#"
"url": "https://www.facebook.com/watch/live/?ref=live_delegate"
},
{
"name": "LiveLeak",
@@ -1983,7 +1983,7 @@
{
"name": "yasiv-youtube",
"type": "url",
"url": "https://yasiv.com/youtube"
"url": "https://yasiv.com/youtube/"
}
]
}
@@ -2101,7 +2101,7 @@
{
"name": "What The Font",
"type": "url",
"url": "https://www.myfonts.com/WhatTheFont/"
"url": "https://www.myfonts.com/pages/whatthefont"
},
{
"name": "Font Squirrel",
@@ -2175,7 +2175,7 @@
{
"name": "Inflact Instagram Viewer (Anonymous)",
"type": "url",
"url": "https://inflact.com/profiles/instagram-viewer/"
"url": "https://inflact.com/instagram-viewer/profile/"
},
{
"name": "Osintgram (T)",
@@ -2345,7 +2345,7 @@
{
"name": "TweetVacuum (T)",
"type": "url",
"url": "https://github.com/T3hUb3rK1tten/TweetVacuum"
"url": "https://github.com/UberKitten/TweetVacuum"
}
]
}
@@ -2358,7 +2358,7 @@
{
"name": "Reddit Metis",
"type": "url",
"url": "https://redditmetis.com/#"
"url": "https://redditmetis.com/"
},
{
"name": "Reddit Archive",
@@ -2384,7 +2384,7 @@
{
"name": "LinkedInt - LinkedIn Recon Tool (T)",
"type": "url",
"url": "https://github.com/vysec/LinkedInt"
"url": "https://github.com/vysecurity/LinkedInt"
},
{
"name": "ScrapedIn (T)",
@@ -2394,7 +2394,7 @@
{
"name": "InSpy (T)",
"type": "url",
"url": "https://github.com/leapsecurity/InSpy"
"url": "https://github.com/jobroche/InSpy"
},
{
"name": "raven (T)",
@@ -2811,7 +2811,7 @@
{
"name": "Registry Finder",
"type": "url",
"url": "https://www.registryfinder.com/"
"url": "https://registryfinder.com:443/"
},
{
"name": "My Registry",
@@ -3002,7 +3002,7 @@
{
"name": "HLR Lookup Portal (R)",
"type": "url",
"url": "https://www.hlr-lookups.com/"
"url": "https://www.hlr-lookups.com/en/start"
},
{
"name": "OpenCNAM API",
@@ -3012,7 +3012,7 @@
{
"name": "Numspy (T)",
"type": "python3 Module",
"url": "https://bhattsameer.github.io/numspy"
"url": "https://bhattsameer.github.io/numspy/"
},
{
"name": "Numspy-Api",
@@ -3113,7 +3113,7 @@
{
"name": "National Sex Offender Search",
"type": "url",
"url": "https://www.nsopw.gov/en/Search"
"url": "https://www.nsopw.gov/search-public-sex-offender-registries"
},
{
"name": "Mugshots.com",
@@ -3280,7 +3280,7 @@
{
"name": "Google Patent Search",
"type": "url",
"url": "https://www.google.com/advanced_patent_search"
"url": "https://patents.google.com/advanced"
}
]
},
@@ -3331,12 +3331,12 @@
"url": "https://www.brbpub.com/"
},
{
"name": "GOVDATA - Das Datenportal f\u00fcr Deutschland (German)",
"name": "GOVDATA - Das Datenportal für Deutschland (German)",
"type": "url",
"url": "https://www.govdata.de/"
},
{
"name": "Open-Data-Portal M\u00fcnchen (German)",
"name": "Open-Data-Portal München (German)",
"type": "url",
"url": "https://www.opengov-muenchen.de/"
},
@@ -3420,7 +3420,7 @@
{
"name": "Google Finance",
"type": "url",
"url": "https://www.google.com/finance"
"url": "https://www.google.com/finance/"
}
]
},
@@ -3471,7 +3471,7 @@
{
"name": "Owler (R)",
"type": "url",
"url": "https://www.owler.com/"
"url": "https://www.owler.com/corp"
},
{
"name": "Vault",
@@ -3741,12 +3741,12 @@
{
"name": "Batch Geocoding",
"type": "url",
"url": "https://www.doogal.co.uk/BatchGeocoding.php"
"url": "https://www.doogal.co.uk/BatchGeocoding"
},
{
"name": "Batch Reverse Geocoding",
"type": "url",
"url": "https://www.doogal.co.uk/BatchReverseGeocoding.php"
"url": "https://www.doogal.co.uk/BatchReverseGeocoding"
}
]
},
@@ -3788,7 +3788,7 @@
{
"name": "OpenSignal",
"type": "url",
"url": "https://opensignal.com/"
"url": "https://www.opensignal.com/"
},
{
"name": "AntennaSearch",
@@ -3870,7 +3870,7 @@
{
"name": "Google Earth",
"type": "url",
"url": "https://www.google.com/earth/"
"url": "https://earth.google.com/web/"
},
{
"name": "Baidu Maps",
@@ -4000,7 +4000,7 @@
{
"name": "Yandex",
"type": "url",
"url": "https://www.yandex.com/"
"url": "https://yandex.com/"
},
{
"name": "Baidu",
@@ -4040,7 +4040,7 @@
{
"name": "Hulbee",
"type": "url",
"url": "https://hulbee.com/"
"url": "https://hulbee.com/de"
},
{
"name": "Mojeek",
@@ -4050,7 +4050,7 @@
{
"name": "Swisscows",
"type": "url",
"url": "https://swisscows.com/"
"url": "https://swisscows.com/en"
},
{
"name": "Brave",
@@ -4117,7 +4117,7 @@
{
"name": "NerdyData",
"type": "url",
"url": "https://nerdydata.com/search"
"url": "https://www.nerdydata.com/reports/new"
},
{
"name": "Gitrob (T)",
@@ -4132,7 +4132,7 @@
{
"name": "GitLeaks",
"type": "url",
"url": "https://github.com/zricethezav/gitleaks"
"url": "https://github.com/gitleaks/gitleaks"
}
]
},
@@ -4325,7 +4325,7 @@
{
"name": "Inshorts",
"type": "url",
"url": "https://www.inshorts.com/en/read"
"url": "https://inshorts.com/en/read"
},
{
"name": "NewsBot",
@@ -4382,7 +4382,7 @@
{
"name": "Google Alerts",
"type": "url",
"url": "https://www.google.com/alerts#"
"url": "https://www.google.com/alerts"
},
{
"name": "Google Custom Search Engine",
@@ -4423,7 +4423,7 @@
{
"name": "Google Hacking Database",
"type": "url",
"url": "https://www.exploit-db.com/google-hacking-database/"
"url": "https://www.exploit-db.com/google-hacking-database"
},
{
"name": "Google Search Operators Guide",
@@ -4575,7 +4575,7 @@
{
"name": "Internet Archive: Wayback Machine",
"type": "url",
"url": "https://archive.org/web/"
"url": "https://web.archive.org/"
},
{
"name": "Archive.is",
@@ -4692,7 +4692,7 @@
{
"name": "UCI Spambase Data Set",
"type": "url",
"url": "https://archive.ics.uci.edu/ml/datasets/Spambase"
"url": "https://archive.ics.uci.edu/dataset/94/spambase"
},
{
"name": "Stanford Large Network Dataset Collection",
@@ -4725,7 +4725,7 @@
{
"name": "DeepL Translator",
"type": "url",
"url": "https://www.deepl.com/"
"url": "https://www.deepl.com/en"
},
{
"name": "Google Translate",
@@ -4817,7 +4817,7 @@
{
"name": "WhatTheFont",
"type": "url",
"url": "https://www.myfonts.com/WhatTheFont/"
"url": "https://www.myfonts.com/pages/whatthefont"
}
]
}
@@ -5067,17 +5067,17 @@
{
"name": "Reddit Deep Web",
"type": "url",
"url": "https://www.reddit.com/r/deepweb"
"url": "https://www.reddit.com/r/deepweb/"
},
{
"name": "Reddit Onions",
"type": "url",
"url": "https://www.reddit.com/r/onions"
"url": "https://www.reddit.com/r/onions/"
},
{
"name": "Reddit Darknet",
"type": "url",
"url": "https://www.reddit.com/r/darknet"
"url": "https://www.reddit.com/r/darknet/"
}
]
},
@@ -5088,7 +5088,7 @@
{
"name": "Tor Download (T)",
"type": "url",
"url": "https://www.torproject.org/download/download-easy.html.en"
"url": "https://www.torproject.org/download/"
},
{
"name": "Freenet Project (T)",
@@ -5943,7 +5943,7 @@
{
"name": "VirusTotal",
"type": "url",
"url": "https://www.virustotal.com/"
"url": "https://www.virustotal.com/gui/"
},
{
"name": "OPSWAT Meta Defender",
@@ -5953,7 +5953,7 @@
{
"name": "Hybrid Analysis",
"type": "url",
"url": "https://www.hybrid-analysis.com/"
"url": "https://hybrid-analysis.com/"
},
{
"name": "Malware Config",
@@ -6083,7 +6083,7 @@
{
"name": "Default Passwords DB",
"type": "url",
"url": "https://cirt.net/passwords"
"url": "https://cirt.net/passwords/"
},
{
"name": "Default passwords list",
@@ -6180,7 +6180,7 @@
{
"name": "Canadian Centre for Cyber Security",
"type": "url",
"url": "https://cyber.gc.ca/"
"url": "https://www.cyber.gc.ca/"
}
]
},
@@ -6246,7 +6246,7 @@
{
"name": "iocextract (T)",
"type": "url",
"url": "https://github.com/InQuest/python-iocextract"
"url": "https://github.com/InQuest/iocextract"
},
{
"name": "ThreatIngestor (T)",
@@ -6359,7 +6359,7 @@
{
"name": "Malware Patrol",
"type": "url",
"url": "https://www.malwarepatrol.net/open-source.shtml"
"url": "https://www.malwarepatrol.net/integrations-formats-threat-intelligence-feed-integration/"
},
{
"name": "AlienVault OTX",
@@ -6500,7 +6500,7 @@
{
"name": "Tor Download (T)",
"type": "url",
"url": "https://www.torproject.org/download/download-easy.html.en"
"url": "https://www.torproject.org/download/"
},
{
"name": "Freenet Project (T)",
@@ -6726,7 +6726,7 @@
"url": "https://themanyhats.club/centralised-place-for-privacy-resources/"
},
{
"name": "The Hitchhiker\u2019s Guide to Online Anonymity",
"name": "The Hitchhikers Guide to Online Anonymity",
"type": "url",
"url": "https://anonymousplanet.org/guide/"
},
@@ -6791,7 +6791,7 @@
{
"name": "Web Page Saver",
"type": "url",
"url": "https://www.magnetforensics.com/free-tool-web-page-saver/"
"url": "https://www.magnetforensics.com/resources/web-page-saver/"
},
{
"name": "Snapper (T)",
@@ -6865,7 +6865,7 @@
{
"name": "GeoGuesser",
"type": "url",
"url": "https://geoguessr.com/"
"url": "https://www.geoguessr.com/"
},
{
"name": "Verif!cation Quiz Bot",
+270
View File
@@ -0,0 +1,270 @@
#!/usr/bin/env python3
"""Apply safe redirect URL updates from link_audit_report.json to arf.json (THE-8)."""
import json
import re
from urllib.parse import urlparse
from pathlib import Path
SCRIPT_DIR = Path(__file__).parent
REPORT_PATH = SCRIPT_DIR / "link_audit_report.json"
ARF_PATH = SCRIPT_DIR.parent / "public" / "arf.json"
# Parked/spam domain prefixes
PARKED_PREFIXES = ("ww1.", "ww2.", "ww3.", "ww4.", "ww5.", "ww7.", "ww12.", "ww25.", "ww38.")
# Login/auth path indicators
LOGIN_INDICATORS = ("login", "signin", "sign-in", "sign_in", "auth", "sso", "oauth", "cas/login", "saml",
"challenge")
# Known paywall/subscription indicators in path
PAYWALL_INDICATORS = ("subscribe", "subscription", "pricing", "plans", "checkout", "purchase", "paywall")
# Specific redirects to skip (manual review needed - destination is wrong/degraded)
MANUAL_SKIP = {
# Specific tool page -> generic corporate homepage
"https://academic.microsoft.com/": "redirects to generic microsoft.com homepage",
# Specific page -> unrelated generic page
"https://getfirebug.com/downloads/": "Firebug is discontinued, redirect to index is not useful",
# Mobile redirect
"https://vk.com/": "redirects to mobile site m.vk.com instead of desktop",
# thatsthem challenge pages
"https://thatsthem.com/name-address-search": "redirects to challenge/captcha gate",
"https://thatsthem.com/reverse-email-lookup": "redirects to challenge/captcha gate",
# Google News advanced search -> generic homepage
"https://news.google.com/news/advanced_news_search?": "advanced search removed, redirects to generic homepage",
# Portswigger duplicate - both URLs go to same place
"https://portswigger.net/burp/download.html": "same destination as existing /burp entry",
# secai domain change that needs verification
"https://secai.ai/research": "subdomain change to i.secai.ai needs verification",
}
def extract_domain(url):
"""Extract the registered domain (without subdomains like www)."""
parsed = urlparse(url)
hostname = parsed.hostname or ""
# Strip www. prefix for comparison
if hostname.startswith("www."):
hostname = hostname[4:]
return hostname
def is_parked_domain(url):
"""Check if URL points to a parked/spam domain."""
parsed = urlparse(url)
hostname = parsed.hostname or ""
return any(hostname.startswith(p) for p in PARKED_PREFIXES)
def is_login_page(url):
"""Check if URL points to a login/auth page."""
parsed = urlparse(url)
path = (parsed.path + "?" + parsed.query).lower()
return any(indicator in path for indicator in LOGIN_INDICATORS)
def is_paywall_page(url):
"""Check if URL points to a paywall/subscription page."""
parsed = urlparse(url)
path = parsed.path.lower()
return any(indicator in path for indicator in PAYWALL_INDICATORS)
def is_generic_homepage_redirect(original_url, final_url):
"""Check if a specific page redirects to a generic homepage on a different path."""
orig = urlparse(original_url)
final = urlparse(final_url)
# Same domain check - if original had a specific path but final is just /
orig_path = orig.path.rstrip("/")
final_path = final.path.rstrip("/")
# Original had a meaningful path, final is root
if orig_path and len(orig_path) > 1 and (not final_path or final_path == ""):
# Only flag if they share the same base domain
orig_domain = extract_domain(original_url)
final_domain = extract_domain(final_url)
if orig_domain == final_domain:
return True
return False
def is_cross_domain_redirect(original_url, final_url):
"""Check if redirect goes to a completely different domain."""
orig_domain = extract_domain(original_url)
final_domain = extract_domain(final_url)
if not orig_domain or not final_domain:
return True
# Allow same domain
if orig_domain == final_domain:
return False
# Allow common subdomain variations (e.g., app.foo.com -> foo.com or foo.com -> www.foo.com)
if orig_domain.endswith("." + final_domain) or final_domain.endswith("." + orig_domain):
return False
# Different domain entirely
return True
def collect_urls(node):
"""Collect all URLs in the arf.json tree, returning a set."""
urls = set()
if node.get("url"):
urls.add(node["url"])
for child in node.get("children", []):
urls.update(collect_urls(child))
return urls
def replace_url_in_tree(node, old_url, new_url):
"""Replace old_url with new_url in tree. Returns count of replacements."""
count = 0
if node.get("url") == old_url:
node["url"] = new_url
count += 1
for child in node.get("children", []):
count += replace_url_in_tree(child, old_url, new_url)
return count
def main():
with open(REPORT_PATH, "r", encoding="utf-8") as f:
report = json.load(f)
with open(ARF_PATH, "r", encoding="utf-8") as f:
arf = json.load(f)
redirects = report.get("redirected", [])
existing_urls = collect_urls(arf)
updated = []
skipped_not_in_arf = []
skipped_empty = []
skipped_login = []
skipped_paywall = []
skipped_parked = []
skipped_homepage = []
skipped_cross_domain = []
skipped_manual = []
for entry in redirects:
old_url = entry.get("url", "")
final_url = entry.get("final_url", "")
# Skip if old URL not in arf.json
if old_url not in existing_urls:
skipped_not_in_arf.append(old_url)
continue
# Skip manual review entries
if old_url in MANUAL_SKIP:
skipped_manual.append((old_url, final_url, MANUAL_SKIP[old_url]))
continue
# Skip empty/null final_url
if not final_url:
skipped_empty.append((old_url, final_url))
continue
# Skip parked domains
if is_parked_domain(final_url):
skipped_parked.append((old_url, final_url))
continue
# Skip login pages
if is_login_page(final_url):
skipped_login.append((old_url, final_url))
continue
# Skip paywall pages
if is_paywall_page(final_url):
skipped_paywall.append((old_url, final_url))
continue
# Skip generic homepage redirects
if is_generic_homepage_redirect(old_url, final_url):
skipped_homepage.append((old_url, final_url))
continue
# Skip suspicious cross-domain redirects
if is_cross_domain_redirect(old_url, final_url):
skipped_cross_domain.append((old_url, final_url))
continue
# Safe to update
count = replace_url_in_tree(arf, old_url, final_url)
if count > 0:
updated.append((old_url, final_url))
# Write back
with open(ARF_PATH, "w", encoding="utf-8") as f:
json.dump(arf, f, indent=2, ensure_ascii=False)
f.write("\n") # trailing newline
# Summary
print("=" * 70)
print("REDIRECT URL UPDATE SUMMARY (THE-8)")
print("=" * 70)
print(f"\nTotal redirects in report: {len(redirects)}")
print(f"Already applied (not in arf.json): {len(skipped_not_in_arf)}")
print(f"Safe updates applied: {len(updated)}")
print(f"\n--- SKIPPED (needs manual review) ---")
print(f"Manual review entries: {len(skipped_manual)}")
print(f"Empty/null final_url: {len(skipped_empty)}")
print(f"Login/auth pages: {len(skipped_login)}")
print(f"Paywall/subscription: {len(skipped_paywall)}")
print(f"Parked/spam domains: {len(skipped_parked)}")
print(f"Generic homepage redirect: {len(skipped_homepage)}")
print(f"Cross-domain redirect: {len(skipped_cross_domain)}")
if updated:
print(f"\n--- APPLIED UPDATES ({len(updated)}) ---")
for old, new in sorted(updated):
print(f" {old}")
print(f" -> {new}")
if skipped_manual:
print(f"\n--- SKIPPED: Manual review ({len(skipped_manual)}) ---")
for old, new, reason in sorted(skipped_manual):
print(f" {old} -> {new}")
print(f" Reason: {reason}")
if skipped_empty:
print(f"\n--- SKIPPED: Empty final_url ({len(skipped_empty)}) ---")
for old, new in sorted(skipped_empty):
print(f" {old}")
if skipped_login:
print(f"\n--- SKIPPED: Login pages ({len(skipped_login)}) ---")
for old, new in sorted(skipped_login):
print(f" {old} -> {new}")
if skipped_paywall:
print(f"\n--- SKIPPED: Paywall ({len(skipped_paywall)}) ---")
for old, new in sorted(skipped_paywall):
print(f" {old} -> {new}")
if skipped_parked:
print(f"\n--- SKIPPED: Parked domains ({len(skipped_parked)}) ---")
for old, new in sorted(skipped_parked):
print(f" {old} -> {new}")
if skipped_homepage:
print(f"\n--- SKIPPED: Generic homepage ({len(skipped_homepage)}) ---")
for old, new in sorted(skipped_homepage):
print(f" {old} -> {new}")
if skipped_cross_domain:
print(f"\n--- SKIPPED: Cross-domain ({len(skipped_cross_domain)}) ---")
for old, new in sorted(skipped_cross_domain):
print(f" {old} -> {new}")
if __name__ == "__main__":
main()