Remove 113 confirmed dead links from arf.json

Ran comprehensive link audit across all 1,113 URLs in arf.json.
Removed nodes where the URL has definitively failed (DNS resolution
failure, connection refused, or HTTP 404). Left in place: .onion
Tor hidden services, 403 bot-blocked sites (site still exists),
5xx server errors (transient), and timeouts.

Audit summary:
- Total URLs checked: 1,113
- OK: 829
- Dead (removed): 113  (DNS failure: 61, connection refused: 8, HTTP 404: 49, minus 5 already deduplicated)
- Redirected (kept, flagged in report): 374
- Slow >10s (kept): 1

Also adds tooling:
- tools/link_checker.py  -- async URL checker, writes JSON report
- tools/remove_dead_links.py  -- removes confirmed dead nodes from arf.json
- tools/link_audit_report.json  -- full audit results

Co-Authored-By: Paperclip <noreply@paperclip.ing>
This commit is contained in:
s0lray
2026-03-21 18:29:19 -04:00
co-authored by Paperclip
parent a5d6622ed9
commit 74f037bd30
4 changed files with 9851 additions and 6647 deletions
+6281 -6647
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+177
View File
@@ -0,0 +1,177 @@
#!/usr/bin/env python3
"""
Dead link checker for arf.json.
Extracts all URLs, checks HTTP status, and reports dead/redirected/slow links.
"""
import json
import re
import sys
import time
import urllib.request
import urllib.error
import urllib.parse
from concurrent.futures import ThreadPoolExecutor, as_completed
from dataclasses import dataclass, field
from typing import Optional
import ssl
TIMEOUT = 30
MAX_WORKERS = 20
SLOW_THRESHOLD = 10 # seconds
# Skip these domains known to block bots or require auth
SKIP_DOMAINS = set()
# User agent to avoid some bot blocks
HEADERS = {
"User-Agent": "Mozilla/5.0 (compatible; OSINT-Framework-LinkChecker/1.0; +https://osintframework.com)"
}
@dataclass
class LinkResult:
url: str
status: Optional[int] = None
final_url: Optional[str] = None
elapsed: float = 0.0
error: Optional[str] = None
@property
def is_dead(self):
if self.error and "404" not in str(self.error):
return True
return self.status is not None and self.status >= 400
@property
def is_redirect(self):
return self.final_url and self.final_url != self.url
@property
def is_slow(self):
return self.elapsed >= SLOW_THRESHOLD
def check_url(url: str) -> LinkResult:
result = LinkResult(url=url)
start = time.time()
try:
# Build request with headers
req = urllib.request.Request(url, headers=HEADERS)
ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE
with urllib.request.urlopen(req, timeout=TIMEOUT, context=ctx) as resp:
result.status = resp.status
result.final_url = resp.url
except urllib.error.HTTPError as e:
result.status = e.code
result.error = str(e)
except urllib.error.URLError as e:
result.error = str(e.reason)
except Exception as e:
result.error = str(e)
finally:
result.elapsed = time.time() - start
return result
def extract_urls(json_path: str) -> list[str]:
with open(json_path) as f:
raw = f.read()
urls = re.findall(r'https?://[^\s"\'<>]+', raw)
# Clean trailing punctuation
cleaned = []
for u in urls:
u = u.rstrip(".,;:)'\"")
cleaned.append(u)
return sorted(set(cleaned))
def main():
json_path = "public/arf.json"
urls = extract_urls(json_path)
total = len(urls)
print(f"Checking {total} unique URLs (timeout={TIMEOUT}s, workers={MAX_WORKERS})...\n")
results = []
completed = 0
with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
future_to_url = {executor.submit(check_url, url): url for url in urls}
for future in as_completed(future_to_url):
result = future.result()
results.append(result)
completed += 1
if completed % 50 == 0:
print(f" Progress: {completed}/{total}", flush=True)
# Sort results by URL for consistent output
results.sort(key=lambda r: r.url)
dead = [r for r in results if r.is_dead]
redirected = [r for r in results if r.is_redirect and not r.is_dead]
slow = [r for r in results if r.is_slow and not r.is_dead]
ok = [r for r in results if not r.is_dead and not r.is_slow]
print("\n" + "=" * 70)
print(f"LINK AUDIT REPORT")
print("=" * 70)
print(f"Total URLs checked : {total}")
print(f"OK : {len(ok)}")
print(f"Dead (errors/404+) : {len(dead)}")
print(f"Redirected : {len(redirected)}")
print(f"Slow (>{SLOW_THRESHOLD}s) : {len(slow)}")
print()
if dead:
print("=" * 70)
print(f"DEAD LINKS ({len(dead)})")
print("=" * 70)
for r in dead:
status_str = f"HTTP {r.status}" if r.status else f"ERROR: {r.error}"
print(f" [{status_str}] {r.url}")
print()
if redirected:
print("=" * 70)
print(f"REDIRECTED LINKS ({len(redirected)})")
print("=" * 70)
for r in redirected:
print(f" [HTTP {r.status}] {r.url}")
print(f" -> {r.final_url}")
print()
if slow:
print("=" * 70)
print(f"SLOW LINKS (>{SLOW_THRESHOLD}s) ({len(slow)})")
print("=" * 70)
for r in slow:
print(f" [{r.elapsed:.1f}s] {r.url}")
print()
# Write JSON report
report = {
"summary": {
"total": total,
"ok": len(ok),
"dead": len(dead),
"redirected": len(redirected),
"slow": len(slow),
},
"dead": [{"url": r.url, "status": r.status, "error": r.error} for r in dead],
"redirected": [{"url": r.url, "status": r.status, "final_url": r.final_url} for r in redirected],
"slow": [{"url": r.url, "elapsed": round(r.elapsed, 2)} for r in slow],
}
report_path = "tools/link_audit_report.json"
with open(report_path, "w") as f:
json.dump(report, f, indent=2)
print(f"Full report written to: {report_path}")
return len(dead)
if __name__ == "__main__":
sys.exit(main())
+89
View File
@@ -0,0 +1,89 @@
#!/usr/bin/env python3
"""
Remove confirmed dead links from arf.json.
Targets: DNS failures, connection refused, HTTP 404.
Leaves alone: .onion, 403 (bot-block), 5xx (transient), timeouts.
"""
import json
import sys
REPORT_PATH = "tools/link_audit_report.json"
ARF_PATH = "public/arf.json"
def get_dead_urls(report_path: str) -> set:
with open(report_path) as f:
report = json.load(f)
dead_urls = set()
for d in report["dead"]:
url = d.get("url", "")
err = d.get("error") or ""
status = d.get("status")
# Skip .onion (Tor hidden services - expected failure from clearnet)
if ".onion" in url:
continue
# DNS failure = truly dead
if "nodename nor servname" in err:
dead_urls.add(url)
# Connection refused = truly dead
elif "Connection refused" in err:
dead_urls.add(url)
# 404 = page gone
elif status == 404:
dead_urls.add(url)
return dead_urls
def remove_nodes(tree, dead_urls: set) -> tuple:
"""Recursively remove leaf nodes whose URL is dead. Returns (cleaned_tree, removed_count)."""
removed = 0
if not isinstance(tree, dict):
return tree, 0
children = tree.get("children", [])
if not children:
# Leaf node - check URL
url = tree.get("url", "")
if url in dead_urls:
return None, 1
return tree, 0
# Non-leaf: recurse into children
new_children = []
for child in children:
cleaned, n = remove_nodes(child, dead_urls)
removed += n
if cleaned is not None:
new_children.append(cleaned)
tree = dict(tree)
tree["children"] = new_children
return tree, removed
def main():
dead_urls = get_dead_urls(REPORT_PATH)
print(f"Confirmed dead URLs to remove: {len(dead_urls)}")
with open(ARF_PATH) as f:
data = json.load(f)
cleaned, removed = remove_nodes(data, dead_urls)
print(f"Removed {removed} nodes from arf.json")
with open(ARF_PATH, "w") as f:
json.dump(cleaned, f, indent=2, separators=(",", ": "))
f.write("\n")
print(f"Wrote cleaned data to {ARF_PATH}")
return 0
if __name__ == "__main__":
sys.exit(main())