mirror of
https://github.com/lockfale/OSINT-Framework.git
synced 2026-08-22 22:02:23 +02:00
Update redirected URLs to final destinations (THE-8)
Apply remaining redirect URL updates in arf.json where the destination is stable and safe. Skips redirects to login pages, parked domains, and cross-domain redirects that need manual review. Co-Authored-By: Paperclip <noreply@paperclip.ing>
This commit is contained in:
+55
-55
@@ -146,7 +146,7 @@
|
||||
{
|
||||
"name": "OSINT Industries",
|
||||
"type": "url",
|
||||
"url": "https://osint.industries/"
|
||||
"url": "https://www.osint.industries/"
|
||||
},
|
||||
{
|
||||
"name": "theHarvester (T)",
|
||||
@@ -312,12 +312,12 @@
|
||||
{
|
||||
"name": "DNSstuff",
|
||||
"type": "url",
|
||||
"url": "https://tools.dnsstuff.com/"
|
||||
"url": "https://www.dnsstuff.com/freetools"
|
||||
},
|
||||
{
|
||||
"name": "Robtex (R)",
|
||||
"type": "url",
|
||||
"url": "https://www.robtex.com/"
|
||||
"url": "https://robtex.com/"
|
||||
},
|
||||
{
|
||||
"name": "Domaincrawler.com",
|
||||
@@ -830,7 +830,7 @@
|
||||
{
|
||||
"name": "Wappalyzer (T)",
|
||||
"type": "url",
|
||||
"url": "https://wappalyzer.com/"
|
||||
"url": "https://www.wappalyzer.com/"
|
||||
},
|
||||
{
|
||||
"name": "SEMrush",
|
||||
@@ -855,7 +855,7 @@
|
||||
{
|
||||
"name": "Open Site Explorer",
|
||||
"type": "url",
|
||||
"url": "https://moz.com/researchtools/ose/"
|
||||
"url": "https://moz.com/link-explorer"
|
||||
},
|
||||
{
|
||||
"name": "SpyOnWeb",
|
||||
@@ -998,7 +998,7 @@
|
||||
{
|
||||
"name": "Google Trends",
|
||||
"type": "url",
|
||||
"url": "https://www.google.com/trends/"
|
||||
"url": "https://trends.google.com/trends/"
|
||||
}
|
||||
]
|
||||
},
|
||||
@@ -1099,7 +1099,7 @@
|
||||
{
|
||||
"name": "Burp Suite (T)",
|
||||
"type": "url",
|
||||
"url": "https://portswigger.net/burp/"
|
||||
"url": "https://portswigger.net/burp"
|
||||
},
|
||||
{
|
||||
"name": "EyeWitness (T)",
|
||||
@@ -1660,7 +1660,7 @@
|
||||
{
|
||||
"name": "Imgrab",
|
||||
"type": "url",
|
||||
"url": "https://imgrab.com/"
|
||||
"url": "https://www.imgrab.com/"
|
||||
},
|
||||
{
|
||||
"name": "Tofo.me",
|
||||
@@ -1917,7 +1917,7 @@
|
||||
{
|
||||
"name": "Facebook Live Map",
|
||||
"type": "url",
|
||||
"url": "https://www.facebook.com/livemap#"
|
||||
"url": "https://www.facebook.com/watch/live/?ref=live_delegate"
|
||||
},
|
||||
{
|
||||
"name": "LiveLeak",
|
||||
@@ -1983,7 +1983,7 @@
|
||||
{
|
||||
"name": "yasiv-youtube",
|
||||
"type": "url",
|
||||
"url": "https://yasiv.com/youtube"
|
||||
"url": "https://yasiv.com/youtube/"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -2101,7 +2101,7 @@
|
||||
{
|
||||
"name": "What The Font",
|
||||
"type": "url",
|
||||
"url": "https://www.myfonts.com/WhatTheFont/"
|
||||
"url": "https://www.myfonts.com/pages/whatthefont"
|
||||
},
|
||||
{
|
||||
"name": "Font Squirrel",
|
||||
@@ -2175,7 +2175,7 @@
|
||||
{
|
||||
"name": "Inflact Instagram Viewer (Anonymous)",
|
||||
"type": "url",
|
||||
"url": "https://inflact.com/profiles/instagram-viewer/"
|
||||
"url": "https://inflact.com/instagram-viewer/profile/"
|
||||
},
|
||||
{
|
||||
"name": "Osintgram (T)",
|
||||
@@ -2345,7 +2345,7 @@
|
||||
{
|
||||
"name": "TweetVacuum (T)",
|
||||
"type": "url",
|
||||
"url": "https://github.com/T3hUb3rK1tten/TweetVacuum"
|
||||
"url": "https://github.com/UberKitten/TweetVacuum"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -2358,7 +2358,7 @@
|
||||
{
|
||||
"name": "Reddit Metis",
|
||||
"type": "url",
|
||||
"url": "https://redditmetis.com/#"
|
||||
"url": "https://redditmetis.com/"
|
||||
},
|
||||
{
|
||||
"name": "Reddit Archive",
|
||||
@@ -2384,7 +2384,7 @@
|
||||
{
|
||||
"name": "LinkedInt - LinkedIn Recon Tool (T)",
|
||||
"type": "url",
|
||||
"url": "https://github.com/vysec/LinkedInt"
|
||||
"url": "https://github.com/vysecurity/LinkedInt"
|
||||
},
|
||||
{
|
||||
"name": "ScrapedIn (T)",
|
||||
@@ -2394,7 +2394,7 @@
|
||||
{
|
||||
"name": "InSpy (T)",
|
||||
"type": "url",
|
||||
"url": "https://github.com/leapsecurity/InSpy"
|
||||
"url": "https://github.com/jobroche/InSpy"
|
||||
},
|
||||
{
|
||||
"name": "raven (T)",
|
||||
@@ -2811,7 +2811,7 @@
|
||||
{
|
||||
"name": "Registry Finder",
|
||||
"type": "url",
|
||||
"url": "https://www.registryfinder.com/"
|
||||
"url": "https://registryfinder.com:443/"
|
||||
},
|
||||
{
|
||||
"name": "My Registry",
|
||||
@@ -3002,7 +3002,7 @@
|
||||
{
|
||||
"name": "HLR Lookup Portal (R)",
|
||||
"type": "url",
|
||||
"url": "https://www.hlr-lookups.com/"
|
||||
"url": "https://www.hlr-lookups.com/en/start"
|
||||
},
|
||||
{
|
||||
"name": "OpenCNAM API",
|
||||
@@ -3012,7 +3012,7 @@
|
||||
{
|
||||
"name": "Numspy (T)",
|
||||
"type": "python3 Module",
|
||||
"url": "https://bhattsameer.github.io/numspy"
|
||||
"url": "https://bhattsameer.github.io/numspy/"
|
||||
},
|
||||
{
|
||||
"name": "Numspy-Api",
|
||||
@@ -3113,7 +3113,7 @@
|
||||
{
|
||||
"name": "National Sex Offender Search",
|
||||
"type": "url",
|
||||
"url": "https://www.nsopw.gov/en/Search"
|
||||
"url": "https://www.nsopw.gov/search-public-sex-offender-registries"
|
||||
},
|
||||
{
|
||||
"name": "Mugshots.com",
|
||||
@@ -3280,7 +3280,7 @@
|
||||
{
|
||||
"name": "Google Patent Search",
|
||||
"type": "url",
|
||||
"url": "https://www.google.com/advanced_patent_search"
|
||||
"url": "https://patents.google.com/advanced"
|
||||
}
|
||||
]
|
||||
},
|
||||
@@ -3331,12 +3331,12 @@
|
||||
"url": "https://www.brbpub.com/"
|
||||
},
|
||||
{
|
||||
"name": "GOVDATA - Das Datenportal f\u00fcr Deutschland (German)",
|
||||
"name": "GOVDATA - Das Datenportal für Deutschland (German)",
|
||||
"type": "url",
|
||||
"url": "https://www.govdata.de/"
|
||||
},
|
||||
{
|
||||
"name": "Open-Data-Portal M\u00fcnchen (German)",
|
||||
"name": "Open-Data-Portal München (German)",
|
||||
"type": "url",
|
||||
"url": "https://www.opengov-muenchen.de/"
|
||||
},
|
||||
@@ -3420,7 +3420,7 @@
|
||||
{
|
||||
"name": "Google Finance",
|
||||
"type": "url",
|
||||
"url": "https://www.google.com/finance"
|
||||
"url": "https://www.google.com/finance/"
|
||||
}
|
||||
]
|
||||
},
|
||||
@@ -3471,7 +3471,7 @@
|
||||
{
|
||||
"name": "Owler (R)",
|
||||
"type": "url",
|
||||
"url": "https://www.owler.com/"
|
||||
"url": "https://www.owler.com/corp"
|
||||
},
|
||||
{
|
||||
"name": "Vault",
|
||||
@@ -3741,12 +3741,12 @@
|
||||
{
|
||||
"name": "Batch Geocoding",
|
||||
"type": "url",
|
||||
"url": "https://www.doogal.co.uk/BatchGeocoding.php"
|
||||
"url": "https://www.doogal.co.uk/BatchGeocoding"
|
||||
},
|
||||
{
|
||||
"name": "Batch Reverse Geocoding",
|
||||
"type": "url",
|
||||
"url": "https://www.doogal.co.uk/BatchReverseGeocoding.php"
|
||||
"url": "https://www.doogal.co.uk/BatchReverseGeocoding"
|
||||
}
|
||||
]
|
||||
},
|
||||
@@ -3788,7 +3788,7 @@
|
||||
{
|
||||
"name": "OpenSignal",
|
||||
"type": "url",
|
||||
"url": "https://opensignal.com/"
|
||||
"url": "https://www.opensignal.com/"
|
||||
},
|
||||
{
|
||||
"name": "AntennaSearch",
|
||||
@@ -3870,7 +3870,7 @@
|
||||
{
|
||||
"name": "Google Earth",
|
||||
"type": "url",
|
||||
"url": "https://www.google.com/earth/"
|
||||
"url": "https://earth.google.com/web/"
|
||||
},
|
||||
{
|
||||
"name": "Baidu Maps",
|
||||
@@ -4000,7 +4000,7 @@
|
||||
{
|
||||
"name": "Yandex",
|
||||
"type": "url",
|
||||
"url": "https://www.yandex.com/"
|
||||
"url": "https://yandex.com/"
|
||||
},
|
||||
{
|
||||
"name": "Baidu",
|
||||
@@ -4040,7 +4040,7 @@
|
||||
{
|
||||
"name": "Hulbee",
|
||||
"type": "url",
|
||||
"url": "https://hulbee.com/"
|
||||
"url": "https://hulbee.com/de"
|
||||
},
|
||||
{
|
||||
"name": "Mojeek",
|
||||
@@ -4050,7 +4050,7 @@
|
||||
{
|
||||
"name": "Swisscows",
|
||||
"type": "url",
|
||||
"url": "https://swisscows.com/"
|
||||
"url": "https://swisscows.com/en"
|
||||
},
|
||||
{
|
||||
"name": "Brave",
|
||||
@@ -4117,7 +4117,7 @@
|
||||
{
|
||||
"name": "NerdyData",
|
||||
"type": "url",
|
||||
"url": "https://nerdydata.com/search"
|
||||
"url": "https://www.nerdydata.com/reports/new"
|
||||
},
|
||||
{
|
||||
"name": "Gitrob (T)",
|
||||
@@ -4132,7 +4132,7 @@
|
||||
{
|
||||
"name": "GitLeaks",
|
||||
"type": "url",
|
||||
"url": "https://github.com/zricethezav/gitleaks"
|
||||
"url": "https://github.com/gitleaks/gitleaks"
|
||||
}
|
||||
]
|
||||
},
|
||||
@@ -4325,7 +4325,7 @@
|
||||
{
|
||||
"name": "Inshorts",
|
||||
"type": "url",
|
||||
"url": "https://www.inshorts.com/en/read"
|
||||
"url": "https://inshorts.com/en/read"
|
||||
},
|
||||
{
|
||||
"name": "NewsBot",
|
||||
@@ -4382,7 +4382,7 @@
|
||||
{
|
||||
"name": "Google Alerts",
|
||||
"type": "url",
|
||||
"url": "https://www.google.com/alerts#"
|
||||
"url": "https://www.google.com/alerts"
|
||||
},
|
||||
{
|
||||
"name": "Google Custom Search Engine",
|
||||
@@ -4423,7 +4423,7 @@
|
||||
{
|
||||
"name": "Google Hacking Database",
|
||||
"type": "url",
|
||||
"url": "https://www.exploit-db.com/google-hacking-database/"
|
||||
"url": "https://www.exploit-db.com/google-hacking-database"
|
||||
},
|
||||
{
|
||||
"name": "Google Search Operators Guide",
|
||||
@@ -4575,7 +4575,7 @@
|
||||
{
|
||||
"name": "Internet Archive: Wayback Machine",
|
||||
"type": "url",
|
||||
"url": "https://archive.org/web/"
|
||||
"url": "https://web.archive.org/"
|
||||
},
|
||||
{
|
||||
"name": "Archive.is",
|
||||
@@ -4692,7 +4692,7 @@
|
||||
{
|
||||
"name": "UCI Spambase Data Set",
|
||||
"type": "url",
|
||||
"url": "https://archive.ics.uci.edu/ml/datasets/Spambase"
|
||||
"url": "https://archive.ics.uci.edu/dataset/94/spambase"
|
||||
},
|
||||
{
|
||||
"name": "Stanford Large Network Dataset Collection",
|
||||
@@ -4725,7 +4725,7 @@
|
||||
{
|
||||
"name": "DeepL Translator",
|
||||
"type": "url",
|
||||
"url": "https://www.deepl.com/"
|
||||
"url": "https://www.deepl.com/en"
|
||||
},
|
||||
{
|
||||
"name": "Google Translate",
|
||||
@@ -4817,7 +4817,7 @@
|
||||
{
|
||||
"name": "WhatTheFont",
|
||||
"type": "url",
|
||||
"url": "https://www.myfonts.com/WhatTheFont/"
|
||||
"url": "https://www.myfonts.com/pages/whatthefont"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -5067,17 +5067,17 @@
|
||||
{
|
||||
"name": "Reddit Deep Web",
|
||||
"type": "url",
|
||||
"url": "https://www.reddit.com/r/deepweb"
|
||||
"url": "https://www.reddit.com/r/deepweb/"
|
||||
},
|
||||
{
|
||||
"name": "Reddit Onions",
|
||||
"type": "url",
|
||||
"url": "https://www.reddit.com/r/onions"
|
||||
"url": "https://www.reddit.com/r/onions/"
|
||||
},
|
||||
{
|
||||
"name": "Reddit Darknet",
|
||||
"type": "url",
|
||||
"url": "https://www.reddit.com/r/darknet"
|
||||
"url": "https://www.reddit.com/r/darknet/"
|
||||
}
|
||||
]
|
||||
},
|
||||
@@ -5088,7 +5088,7 @@
|
||||
{
|
||||
"name": "Tor Download (T)",
|
||||
"type": "url",
|
||||
"url": "https://www.torproject.org/download/download-easy.html.en"
|
||||
"url": "https://www.torproject.org/download/"
|
||||
},
|
||||
{
|
||||
"name": "Freenet Project (T)",
|
||||
@@ -5943,7 +5943,7 @@
|
||||
{
|
||||
"name": "VirusTotal",
|
||||
"type": "url",
|
||||
"url": "https://www.virustotal.com/"
|
||||
"url": "https://www.virustotal.com/gui/"
|
||||
},
|
||||
{
|
||||
"name": "OPSWAT Meta Defender",
|
||||
@@ -5953,7 +5953,7 @@
|
||||
{
|
||||
"name": "Hybrid Analysis",
|
||||
"type": "url",
|
||||
"url": "https://www.hybrid-analysis.com/"
|
||||
"url": "https://hybrid-analysis.com/"
|
||||
},
|
||||
{
|
||||
"name": "Malware Config",
|
||||
@@ -6083,7 +6083,7 @@
|
||||
{
|
||||
"name": "Default Passwords DB",
|
||||
"type": "url",
|
||||
"url": "https://cirt.net/passwords"
|
||||
"url": "https://cirt.net/passwords/"
|
||||
},
|
||||
{
|
||||
"name": "Default passwords list",
|
||||
@@ -6180,7 +6180,7 @@
|
||||
{
|
||||
"name": "Canadian Centre for Cyber Security",
|
||||
"type": "url",
|
||||
"url": "https://cyber.gc.ca/"
|
||||
"url": "https://www.cyber.gc.ca/"
|
||||
}
|
||||
]
|
||||
},
|
||||
@@ -6246,7 +6246,7 @@
|
||||
{
|
||||
"name": "iocextract (T)",
|
||||
"type": "url",
|
||||
"url": "https://github.com/InQuest/python-iocextract"
|
||||
"url": "https://github.com/InQuest/iocextract"
|
||||
},
|
||||
{
|
||||
"name": "ThreatIngestor (T)",
|
||||
@@ -6359,7 +6359,7 @@
|
||||
{
|
||||
"name": "Malware Patrol",
|
||||
"type": "url",
|
||||
"url": "https://www.malwarepatrol.net/open-source.shtml"
|
||||
"url": "https://www.malwarepatrol.net/integrations-formats-threat-intelligence-feed-integration/"
|
||||
},
|
||||
{
|
||||
"name": "AlienVault OTX",
|
||||
@@ -6500,7 +6500,7 @@
|
||||
{
|
||||
"name": "Tor Download (T)",
|
||||
"type": "url",
|
||||
"url": "https://www.torproject.org/download/download-easy.html.en"
|
||||
"url": "https://www.torproject.org/download/"
|
||||
},
|
||||
{
|
||||
"name": "Freenet Project (T)",
|
||||
@@ -6726,7 +6726,7 @@
|
||||
"url": "https://themanyhats.club/centralised-place-for-privacy-resources/"
|
||||
},
|
||||
{
|
||||
"name": "The Hitchhiker\u2019s Guide to Online Anonymity",
|
||||
"name": "The Hitchhiker’s Guide to Online Anonymity",
|
||||
"type": "url",
|
||||
"url": "https://anonymousplanet.org/guide/"
|
||||
},
|
||||
@@ -6791,7 +6791,7 @@
|
||||
{
|
||||
"name": "Web Page Saver",
|
||||
"type": "url",
|
||||
"url": "https://www.magnetforensics.com/free-tool-web-page-saver/"
|
||||
"url": "https://www.magnetforensics.com/resources/web-page-saver/"
|
||||
},
|
||||
{
|
||||
"name": "Snapper (T)",
|
||||
@@ -6865,7 +6865,7 @@
|
||||
{
|
||||
"name": "GeoGuesser",
|
||||
"type": "url",
|
||||
"url": "https://geoguessr.com/"
|
||||
"url": "https://www.geoguessr.com/"
|
||||
},
|
||||
{
|
||||
"name": "Verif!cation Quiz Bot",
|
||||
|
||||
@@ -0,0 +1,270 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Apply safe redirect URL updates from link_audit_report.json to arf.json (THE-8)."""
|
||||
|
||||
import json
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
from pathlib import Path
|
||||
|
||||
SCRIPT_DIR = Path(__file__).parent
|
||||
REPORT_PATH = SCRIPT_DIR / "link_audit_report.json"
|
||||
ARF_PATH = SCRIPT_DIR.parent / "public" / "arf.json"
|
||||
|
||||
# Parked/spam domain prefixes
|
||||
PARKED_PREFIXES = ("ww1.", "ww2.", "ww3.", "ww4.", "ww5.", "ww7.", "ww12.", "ww25.", "ww38.")
|
||||
|
||||
# Login/auth path indicators
|
||||
LOGIN_INDICATORS = ("login", "signin", "sign-in", "sign_in", "auth", "sso", "oauth", "cas/login", "saml",
|
||||
"challenge")
|
||||
|
||||
# Known paywall/subscription indicators in path
|
||||
PAYWALL_INDICATORS = ("subscribe", "subscription", "pricing", "plans", "checkout", "purchase", "paywall")
|
||||
|
||||
# Specific redirects to skip (manual review needed - destination is wrong/degraded)
|
||||
MANUAL_SKIP = {
|
||||
# Specific tool page -> generic corporate homepage
|
||||
"https://academic.microsoft.com/": "redirects to generic microsoft.com homepage",
|
||||
# Specific page -> unrelated generic page
|
||||
"https://getfirebug.com/downloads/": "Firebug is discontinued, redirect to index is not useful",
|
||||
# Mobile redirect
|
||||
"https://vk.com/": "redirects to mobile site m.vk.com instead of desktop",
|
||||
# thatsthem challenge pages
|
||||
"https://thatsthem.com/name-address-search": "redirects to challenge/captcha gate",
|
||||
"https://thatsthem.com/reverse-email-lookup": "redirects to challenge/captcha gate",
|
||||
# Google News advanced search -> generic homepage
|
||||
"https://news.google.com/news/advanced_news_search?": "advanced search removed, redirects to generic homepage",
|
||||
# Portswigger duplicate - both URLs go to same place
|
||||
"https://portswigger.net/burp/download.html": "same destination as existing /burp entry",
|
||||
# secai domain change that needs verification
|
||||
"https://secai.ai/research": "subdomain change to i.secai.ai needs verification",
|
||||
}
|
||||
|
||||
|
||||
def extract_domain(url):
|
||||
"""Extract the registered domain (without subdomains like www)."""
|
||||
parsed = urlparse(url)
|
||||
hostname = parsed.hostname or ""
|
||||
# Strip www. prefix for comparison
|
||||
if hostname.startswith("www."):
|
||||
hostname = hostname[4:]
|
||||
return hostname
|
||||
|
||||
|
||||
def is_parked_domain(url):
|
||||
"""Check if URL points to a parked/spam domain."""
|
||||
parsed = urlparse(url)
|
||||
hostname = parsed.hostname or ""
|
||||
return any(hostname.startswith(p) for p in PARKED_PREFIXES)
|
||||
|
||||
|
||||
def is_login_page(url):
|
||||
"""Check if URL points to a login/auth page."""
|
||||
parsed = urlparse(url)
|
||||
path = (parsed.path + "?" + parsed.query).lower()
|
||||
return any(indicator in path for indicator in LOGIN_INDICATORS)
|
||||
|
||||
|
||||
def is_paywall_page(url):
|
||||
"""Check if URL points to a paywall/subscription page."""
|
||||
parsed = urlparse(url)
|
||||
path = parsed.path.lower()
|
||||
return any(indicator in path for indicator in PAYWALL_INDICATORS)
|
||||
|
||||
|
||||
def is_generic_homepage_redirect(original_url, final_url):
|
||||
"""Check if a specific page redirects to a generic homepage on a different path."""
|
||||
orig = urlparse(original_url)
|
||||
final = urlparse(final_url)
|
||||
|
||||
# Same domain check - if original had a specific path but final is just /
|
||||
orig_path = orig.path.rstrip("/")
|
||||
final_path = final.path.rstrip("/")
|
||||
|
||||
# Original had a meaningful path, final is root
|
||||
if orig_path and len(orig_path) > 1 and (not final_path or final_path == ""):
|
||||
# Only flag if they share the same base domain
|
||||
orig_domain = extract_domain(original_url)
|
||||
final_domain = extract_domain(final_url)
|
||||
if orig_domain == final_domain:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def is_cross_domain_redirect(original_url, final_url):
|
||||
"""Check if redirect goes to a completely different domain."""
|
||||
orig_domain = extract_domain(original_url)
|
||||
final_domain = extract_domain(final_url)
|
||||
|
||||
if not orig_domain or not final_domain:
|
||||
return True
|
||||
|
||||
# Allow same domain
|
||||
if orig_domain == final_domain:
|
||||
return False
|
||||
|
||||
# Allow common subdomain variations (e.g., app.foo.com -> foo.com or foo.com -> www.foo.com)
|
||||
if orig_domain.endswith("." + final_domain) or final_domain.endswith("." + orig_domain):
|
||||
return False
|
||||
|
||||
# Different domain entirely
|
||||
return True
|
||||
|
||||
|
||||
def collect_urls(node):
|
||||
"""Collect all URLs in the arf.json tree, returning a set."""
|
||||
urls = set()
|
||||
if node.get("url"):
|
||||
urls.add(node["url"])
|
||||
for child in node.get("children", []):
|
||||
urls.update(collect_urls(child))
|
||||
return urls
|
||||
|
||||
|
||||
def replace_url_in_tree(node, old_url, new_url):
|
||||
"""Replace old_url with new_url in tree. Returns count of replacements."""
|
||||
count = 0
|
||||
if node.get("url") == old_url:
|
||||
node["url"] = new_url
|
||||
count += 1
|
||||
for child in node.get("children", []):
|
||||
count += replace_url_in_tree(child, old_url, new_url)
|
||||
return count
|
||||
|
||||
|
||||
def main():
|
||||
with open(REPORT_PATH, "r", encoding="utf-8") as f:
|
||||
report = json.load(f)
|
||||
|
||||
with open(ARF_PATH, "r", encoding="utf-8") as f:
|
||||
arf = json.load(f)
|
||||
|
||||
redirects = report.get("redirected", [])
|
||||
existing_urls = collect_urls(arf)
|
||||
|
||||
updated = []
|
||||
skipped_not_in_arf = []
|
||||
skipped_empty = []
|
||||
skipped_login = []
|
||||
skipped_paywall = []
|
||||
skipped_parked = []
|
||||
skipped_homepage = []
|
||||
skipped_cross_domain = []
|
||||
skipped_manual = []
|
||||
|
||||
for entry in redirects:
|
||||
old_url = entry.get("url", "")
|
||||
final_url = entry.get("final_url", "")
|
||||
|
||||
# Skip if old URL not in arf.json
|
||||
if old_url not in existing_urls:
|
||||
skipped_not_in_arf.append(old_url)
|
||||
continue
|
||||
|
||||
# Skip manual review entries
|
||||
if old_url in MANUAL_SKIP:
|
||||
skipped_manual.append((old_url, final_url, MANUAL_SKIP[old_url]))
|
||||
continue
|
||||
|
||||
# Skip empty/null final_url
|
||||
if not final_url:
|
||||
skipped_empty.append((old_url, final_url))
|
||||
continue
|
||||
|
||||
# Skip parked domains
|
||||
if is_parked_domain(final_url):
|
||||
skipped_parked.append((old_url, final_url))
|
||||
continue
|
||||
|
||||
# Skip login pages
|
||||
if is_login_page(final_url):
|
||||
skipped_login.append((old_url, final_url))
|
||||
continue
|
||||
|
||||
# Skip paywall pages
|
||||
if is_paywall_page(final_url):
|
||||
skipped_paywall.append((old_url, final_url))
|
||||
continue
|
||||
|
||||
# Skip generic homepage redirects
|
||||
if is_generic_homepage_redirect(old_url, final_url):
|
||||
skipped_homepage.append((old_url, final_url))
|
||||
continue
|
||||
|
||||
# Skip suspicious cross-domain redirects
|
||||
if is_cross_domain_redirect(old_url, final_url):
|
||||
skipped_cross_domain.append((old_url, final_url))
|
||||
continue
|
||||
|
||||
# Safe to update
|
||||
count = replace_url_in_tree(arf, old_url, final_url)
|
||||
if count > 0:
|
||||
updated.append((old_url, final_url))
|
||||
|
||||
# Write back
|
||||
with open(ARF_PATH, "w", encoding="utf-8") as f:
|
||||
json.dump(arf, f, indent=2, ensure_ascii=False)
|
||||
f.write("\n") # trailing newline
|
||||
|
||||
# Summary
|
||||
print("=" * 70)
|
||||
print("REDIRECT URL UPDATE SUMMARY (THE-8)")
|
||||
print("=" * 70)
|
||||
|
||||
print(f"\nTotal redirects in report: {len(redirects)}")
|
||||
print(f"Already applied (not in arf.json): {len(skipped_not_in_arf)}")
|
||||
print(f"Safe updates applied: {len(updated)}")
|
||||
|
||||
print(f"\n--- SKIPPED (needs manual review) ---")
|
||||
print(f"Manual review entries: {len(skipped_manual)}")
|
||||
print(f"Empty/null final_url: {len(skipped_empty)}")
|
||||
print(f"Login/auth pages: {len(skipped_login)}")
|
||||
print(f"Paywall/subscription: {len(skipped_paywall)}")
|
||||
print(f"Parked/spam domains: {len(skipped_parked)}")
|
||||
print(f"Generic homepage redirect: {len(skipped_homepage)}")
|
||||
print(f"Cross-domain redirect: {len(skipped_cross_domain)}")
|
||||
|
||||
if updated:
|
||||
print(f"\n--- APPLIED UPDATES ({len(updated)}) ---")
|
||||
for old, new in sorted(updated):
|
||||
print(f" {old}")
|
||||
print(f" -> {new}")
|
||||
|
||||
if skipped_manual:
|
||||
print(f"\n--- SKIPPED: Manual review ({len(skipped_manual)}) ---")
|
||||
for old, new, reason in sorted(skipped_manual):
|
||||
print(f" {old} -> {new}")
|
||||
print(f" Reason: {reason}")
|
||||
|
||||
if skipped_empty:
|
||||
print(f"\n--- SKIPPED: Empty final_url ({len(skipped_empty)}) ---")
|
||||
for old, new in sorted(skipped_empty):
|
||||
print(f" {old}")
|
||||
|
||||
if skipped_login:
|
||||
print(f"\n--- SKIPPED: Login pages ({len(skipped_login)}) ---")
|
||||
for old, new in sorted(skipped_login):
|
||||
print(f" {old} -> {new}")
|
||||
|
||||
if skipped_paywall:
|
||||
print(f"\n--- SKIPPED: Paywall ({len(skipped_paywall)}) ---")
|
||||
for old, new in sorted(skipped_paywall):
|
||||
print(f" {old} -> {new}")
|
||||
|
||||
if skipped_parked:
|
||||
print(f"\n--- SKIPPED: Parked domains ({len(skipped_parked)}) ---")
|
||||
for old, new in sorted(skipped_parked):
|
||||
print(f" {old} -> {new}")
|
||||
|
||||
if skipped_homepage:
|
||||
print(f"\n--- SKIPPED: Generic homepage ({len(skipped_homepage)}) ---")
|
||||
for old, new in sorted(skipped_homepage):
|
||||
print(f" {old} -> {new}")
|
||||
|
||||
if skipped_cross_domain:
|
||||
print(f"\n--- SKIPPED: Cross-domain ({len(skipped_cross_domain)}) ---")
|
||||
for old, new in sorted(skipped_cross_domain):
|
||||
print(f" {old} -> {new}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user