Parallelize & clean up link checker

This commit is contained in:
mDuo13
2026-06-01 13:32:09 -07:00
parent 0f3b683e1b
commit 25232e8db6
3 changed files with 208 additions and 157 deletions

View File

@@ -62,7 +62,7 @@ jobs:
- name: Run Selenium link checker
run: |
set -eo pipefail
python tools/check-external-links.py --live docs/ 2>&1 | tee selenium-link-check.log
python tools/check-links.py docs/ 2>&1 | tee selenium-link-check.log
- name: Stop Realm server
if: always()

View File

@@ -12,12 +12,14 @@
# * may be used as a wildcard AT THE END OF LINKS ONLY
# These sites block crawlers (often using Cloudflare) with a 403 but typically
# work when accessed from an actual browser:
# work when accessed from an actual browser. It's a good idea to re-check them
# manually every once in a while to make sure they really do still work.
https://x.com/*
https://www.gnu.org/software/screen/manual/screen.html
https://www.coindesk.com/markets/2015/04/02/1-million-legal-fight-ensnares-ripple-bitstamp-and-jed-mccaleb/
https://www.coindesk.com/markets/2016/02/12/ripple-settles-1-million-lawsuit-with-former-executive-and-founder/
https://bsaaml.ffiec.gov/manual/Introduction/01
https://bsaaml.ffiec.gov/manual/AssessingComplianceWithBSARegulatoryRequirements/04
https://www.sec.gov/oiea/investor-alerts-and-bulletins/ib_coinofferings
https://www.npmjs.com/*
https://go.dev/dl/

View File

@@ -7,13 +7,11 @@ tools/check-external-links.py [folder/to/check/]
If [folder/to/check] is omitted, check ./docs/
Prints a report of broken links. In the default mode, checks external links in
Markdown files only. Pass --live to test against a local Redocly dev server,
which checks *all* links, including anchors and markdoc tags; assumes the
Prints a report of broken links. Tests against a local Redocly dev server,
and checks *all* links, including anchors and markdoc tags; assumes the
server is already up and running on localhost:4000.
Requires: beautifulsoup4, markdown, requests
For live mode, selenium and Chrome are also required.
Requires: requests, selenium (w/ Chrome WebDriver)
Links & sites that often report false-positives can be added to broken-links.txt
to have the link checker skip them.
@@ -24,12 +22,11 @@ import json
import logging
import os
import re
import threading
from time import time, sleep
import requests
from requests.adapters import HTTPAdapter, Retry
from markdown import markdown
from bs4 import BeautifulSoup
from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.common.exceptions import StaleElementReferenceException, NoSuchElementException
@@ -45,6 +42,8 @@ DEFAULT_SKIP_PATHS = [
"_code-samples", # Debatably, we might want to link-check the READMEs here
"_api-examples",
"_sources",
"_snippets",
"img",
]
MAX_RETRIES = 1 # Times to retry if a link doesn't work
TIMEOUT_SECONDS = 8 # Seconds before giving up on a link
@@ -63,23 +62,110 @@ logger = logging.getLogger(__name__)
logger.addHandler(logging.StreamHandler())
logger.propagate = False
class CheckerSession:
"""
Instance of all the link-checking tools to be used by a given thread,
including two Chrome webdrivers and two Requests sessions with automatic
backoff and custom user-agent.
"""
def __init__(self):
retries = Retry(total=MAX_RETRIES, backoff_factor=1, status_forcelist=[ 502, 503, 504 ])
self.s = requests.Session()
self.s.mount("https://", HTTPAdapter(max_retries=retries))
self.h = requests.Session()
self.h.mount("http://", HTTPAdapter(max_retries=retries))
self.h.headers.update({"User-Agent": USER_AGENT})
self.s.headers.update({"User-Agent": USER_AGENT})
self.last_host_called = None
options = webdriver.ChromeOptions()
options.add_argument("--headless=new")
self.chrome = webdriver.Chrome(options=options)
# Make a second one so we can check for anchors without resetting
# all the references we have in the page the link comes from
self.chrome2 = webdriver.Chrome(options=options)
def fetch_code(self, href: str):
"""
Get status code of a URL, using saved sessions & retries, automatically
failing over from HTTP HEAD to HTTP GET and adding a slight delay to
avoid hammering the same host repeatedly.
"""
proto,predicate = href.split("//",1)
host = predicate.split("/", 1)[0]
if host == self.last_host_called:
sleep(SAME_HOST_DELAY)
self.last_host_called = host
if href.startswith("http://"):
sess = self.h
else:
sess = self.s
try:
code = sess.head(href, timeout=TIMEOUT_SECONDS).status_code
except Exception as e:
logger.debug(f"Error getting {href}: {e}")
try:
code = sess.get(href, timeout=TIMEOUT_SECONDS).status_code
except Exception as e2:
logger.debug(f"Error getting {href}: {e2}")
code = 500
return code
def get_page_hrefs_and_text(self, href: str):
self.chrome.get(href)
rootlayout = self.chrome.find_element(By.CSS_SELECTOR, '[data-component-name="layouts/RootLayout"]')
try:
links = rootlayout.find_elements(By.CSS_SELECTOR, "a")
pagetext = rootlayout.text
hrefs = [link.get_attribute("href") for link in links]
except StaleElementReferenceException:
# This can happen when hydration fails or the page is updated
# asynchronously, for example by the amendment-disclaimer tag.
# Try again and hopefully it works this time.
sleep(0.2)
rootlayout = self.chrome.find_element(By.CSS_SELECTOR, '[data-component-name="layouts/RootLayout"]')
pagetext = rootlayout.text
links = rootlayout.find_elements(By.CSS_SELECTOR, "a")
hrefs = [link.get_attribute("href") for link in links]
return hrefs, pagetext
def find_id_in_page(self, href: str, id: str):
"""
Return the element with a given unique ID in the DOM, or None if no
matching element is found. Uses the second chrome driver to avoid
messing with the first driver's view of the page the link is from
"""
self.chrome2.get(href)
sleep(0.3) # delay to give it time for hydration failures
try:
el = self.chrome2.find_element(By.ID, id)
except NoSuchElementException:
el = None
return el
class LinkChecker:
def __init__(self, topdir,
skip_paths = DEFAULT_SKIP_PATHS,
cache_file = DEFAULT_CACHE_FILE,
known_broken = KNOWN_BROKEN_LINKS_FILE,
live = False
known_broken = KNOWN_BROKEN_LINKS_FILE
):
self.topdir = topdir
self.skip_paths = skip_paths
self.last_checkin = time()
self.last_cache_update = 0
self.live = live
self.setup_sessions()
self.init_cache(cache_file)
self.init_known_broken(known_broken)
def init_cache(self, cache_file: str):
"""
Set up a local cache of link checker results so multiple links to the
same place don't have to be checked repeatedly. The cache can be saved
to a JSON file and loaded from it to save time across runs, but even if
it isn't, a cache dict is used during a single run to save time.
The cache uses a lock so that multiple threads don't write to it at the
same time, but reading from the cache doesn't require the lock.
"""
self.cache_lock = threading.Lock()
self.anchor_cache = {}
if not cache_file:
logger.debug("No cache file, not loading anything.")
@@ -116,31 +202,23 @@ class LinkChecker:
self.last_cache_update = time()
def write_cache(self):
if not self.cache_file:
return
with open(self.cache_file, "w") as f:
json.dump(self.cache, f)
self.last_cache_update = time()
def setup_sessions(self):
retries = Retry(total=MAX_RETRIES, backoff_factor=1, status_forcelist=[ 502, 503, 504 ])
self.s = requests.Session()
self.s.mount("https://", HTTPAdapter(max_retries=retries))
self.h = requests.Session()
self.h.mount("http://", HTTPAdapter(max_retries=retries))
self.h.headers.update({"User-Agent": USER_AGENT})
self.s.headers.update({"User-Agent": USER_AGENT})
self.last_host_called = None
if self.live:
options = webdriver.ChromeOptions()
options.add_argument("--headless=new")
self.chrome = webdriver.Chrome(options=options)
# Make a second one so we can check for anchors without resetting
# all the references we have in the page the link comes from
self.chrome2 = webdriver.Chrome(options=options)
"""
Update the cache file with the latest of the cache, so even if the run
doesn't finish successfully, some progress is saved.
"""
with self.cache_lock:
if not self.cache_file:
return
with open(self.cache_file, "w") as f:
json.dump(self.cache, f)
self.last_cache_update = time()
def init_known_broken(self, known_broken):
"""
Read the known broken links config file and set up internal structures
for recognizing 'known broken' links that should be excluded from
checking, based on its contents.
"""
self.exact_known_broken = []
self.wildcard_known_broken = []
@@ -174,12 +252,51 @@ class LinkChecker:
self.write_cache()
def walk(self):
logger.info(f"Checking files in {os.path.abspath(self.topdir)}")
externalCache = []
"""
Walk all folders in the top dir and check for
broken links, paralellizing as possible.
"""
threads = []
self.broken_links = []
self.total_links_checked = 0
self.report_lock = threading.Lock()
dirnames = [d.name for d in os.scandir(self.topdir) if d.is_dir()]
dirnames[:] = [d for d in dirnames if d not in self.skip_paths]
for dirpath in dirnames:
dirpath = os.path.join(self.topdir, dirpath)
t = threading.Thread(target=self.check_dir, args=(dirpath,))
threads.append(t)
top_dir_thread = threading.Thread(target=self.check_dir,
args=(self.topdir, False))
threads.append(top_dir_thread)
logger.info(f"Checking dirs: {'\n '.join(dirnames)}")
logger.info(f"Starting {len(threads)} threads to check links...")
for t in threads:
t.start()
for t in threads:
t.join()
self.report()
def record_broken(self, broken_links: list, num_checked: int):
"""
Helper for checker threads to report to the main thread with their
lists of broken links
"""
with self.report_lock:
self.broken_links += broken_links
self.total_links_checked += num_checked
def check_dir(self, top_dirpath, recurse=True):
"""
Check a single directory (& subdirs) for broken links. Capable of
running in a separate thread.
"""
logger.info(f"Checking files in {os.path.abspath(top_dirpath)}")
broken_links = []
total_links_checked = 0
last_checkin = time()
for dirpath, dirnames, filenames in os.walk(self.topdir):
sess = CheckerSession()
for dirpath, dirnames, filenames in os.walk(top_dirpath):
dirnames[:] = [d for d in dirnames if d not in self.skip_paths]
self.checkin(f"dir: {dirpath}")
if dirpath in self.skip_paths:
@@ -188,51 +305,17 @@ class LinkChecker:
for fname in filenames:
self.checkin(f"file: {fname}")
in_file = os.path.join(dirpath, fname)
if in_file.endswith(".md"):
if self.live:
newly_checked, newly_broken = self.check_file_live(in_file)
else:
newly_checked, newly_broken = self.check_file(in_file)
if in_file.endswith(".md") or in_file.endswith(".page.tsx"):
newly_checked, newly_broken = self.check_file(in_file, sess)
broken_links += newly_broken
total_links_checked += newly_checked
if in_file.endswith(".page.tsx") and self.live:
newly_checked, newly_broken = self.check_file_live(in_file)
broken_links += newly_broken
total_links_checked += newly_checked
self.report(broken_links, total_links_checked)
else:
logger.debug(f"Not checking non-page file {in_file}")
if not recurse:
break
self.record_broken(broken_links, total_links_checked)
def check_file(self, in_file: str):
"""
Given a specific .md file, look for external links in it and check them.
Returns how many were checked and a list of tuples where each member is
a (file,link) pair representing a broken link.
Note that this does not parse Markdoc including partials so it can't
handle the full context of the Redocly parser but those *usually* won't
be needed for external links.
"""
logger.info(f"Checking file {in_file}")
with open(in_file, 'r', encoding="utf-8") as f:
html = markdown(f.read())
soup = BeautifulSoup(html, "html.parser")
links = soup.find_all("a")
broken = []
num_checked = 0
for link in links:
self.checkin(f"link: {link}")
if "href" not in link.attrs:
# probably an <a name> type anchor, skip
continue
if (link["href"].startswith("https://") or
link["href"].startswith("http://")):
was_checked, was_good = self.check_link(link["href"])
num_checked += was_checked
if not was_good:
broken.append( (in_file, link["href"]) )
return num_checked, broken
def check_file_live(self, in_file: str):
def check_file(self, in_file: str, sess: CheckerSession):
"""
Given a specific .md file, fetch it from the Redocly dev server and
check for links in it.
@@ -246,7 +329,7 @@ class LinkChecker:
logger.warning(f"Not checking path that's not an md or page.tsx file: {in_file}")
return (0, [])
url = REDOCLY_DEV_BASE + path
code = self.fetch(url)
code = sess.fetch_code(url)
if code < 200 or code >= 400:
logger.warning(f"Failed to get page from dev server for file {in_file}")
return 0, []
@@ -254,34 +337,19 @@ class LinkChecker:
broken = []
num_checked = 0
logger.info(f"Checking path {path}")
self.chrome.get(REDOCLY_DEV_BASE+path)
# sleep(0.2) # give it a moment in case hydration fails or something
rootlayout = self.chrome.find_element(By.CSS_SELECTOR, '[data-component-name="layouts/RootLayout"]')
try:
links = rootlayout.find_elements(By.CSS_SELECTOR, "a")
pagetext = rootlayout.text
hrefs = [link.get_attribute("href") for link in links]
except StaleElementReferenceException:
# This can happen when hydration fails or the page is updated
# asynchronously, for example by fetching amendment status.
# Try again and hopefully it works this time.
sleep(0.2)
rootlayout = self.chrome.find_element(By.CSS_SELECTOR, '[data-component-name="layouts/RootLayout"]')
pagetext = rootlayout.text
links = rootlayout.find_elements(By.CSS_SELECTOR, "a")
hrefs = [link.get_attribute("href") for link in links]
hrefs, pagetext = sess.get_page_hrefs_and_text(REDOCLY_DEV_BASE+path)
for href in hrefs:
self.checkin(f"link: {href}")
if not href:
# Probably a name anchor something, skip
continue
if href.startswith(REDOCLY_DEV_BASE):
was_checked, was_good = self.check_dev_link(href)
was_checked, was_good = self.check_dev_link(href, sess)
num_checked += was_checked
if not was_good:
broken.append( (in_file, href) )
elif href.startswith("http://") or href.startswith("https://"):
was_checked, was_good = self.check_link(href)
was_checked, was_good = self.check_link(href, sess)
num_checked += was_checked
if not was_good:
broken.append( (in_file, href) )
@@ -291,6 +359,10 @@ class LinkChecker:
return num_checked, broken
def check_for_unparsed_reflinks(self, text: str):
"""
Given text of a page, find patterns like [Hash][] that should have been
parsed into links from the Markdown source, but weren't.
"""
unparsed_links = []
matches = UNMATCHED_REFLINK_REGEX.finditer(text)
for m in matches:
@@ -306,41 +378,22 @@ class LinkChecker:
return True
return False
def fetch(self, href: str):
"""
Get status code of a URL, using saved sessions & retries, automatically
failing over from HTTP HEAD to HTTP GET and adding a slight delay to
avoid hammering the same host repeatedly.
"""
proto,predicate = href.split("//",1)
host = predicate.split("/", 1)[0]
if host == self.last_host_called:
sleep(SAME_HOST_DELAY)
self.last_host_called = host
if href.startswith("http://"):
sess = self.h
else:
sess = self.s
try:
code = sess.head(href, timeout=TIMEOUT_SECONDS).status_code
except Exception as e:
logger.debug(f"Error getting {href}: {e}")
try:
code = sess.get(href, timeout=TIMEOUT_SECONDS).status_code
except Exception as e2:
logger.debug(f"Error getting {href}: {e2}")
code = 500
return code
def trim_trackers(self, href: str):
"""
Remove query parameters that are added by JavaScript trackers. These
parameters tend to include timestamps that bust the cache unnecessarily.
"""
if "?__hstc=" in href:
return href[:href.find("?__hstc=")]
return href
def check_link(self, href: str):
def check_link(self, href: str, sess: CheckerSession):
"""
Check a link to see if it can be successfully fetched. Returns a tuple
(num_checked: int, was_good: bool) where num_checked is 1 if the link
was checked and 0 if it was skipped, and was_good is True if the link
was fetched successfully and False if it failed.
"""
href = self.trim_trackers(href)
if href in self.cache.keys():
logger.debug(f"... Skipping (cached): {href}")
@@ -350,16 +403,17 @@ class LinkChecker:
logger.debug(f"... Skipping (known broken): {href}")
return (0, True)
logger.info(f"... Testing link {href}")
code = self.fetch(href)
code = sess.fetch_code(href)
if code < 200 or code >= 400:
logger.warning(f"... Broken link to {href}")
self.cache[href] = (False, time())
with self.cache_lock:
self.cache[href] = (False, time())
return (1, False)
logger.info("... ... success.")
self.cache[href] = (True, time())
with self.cache_lock:
self.cache[href] = (True, time())
return (1, True)
def check_dev_link(self, href: str):
def check_dev_link(self, href: str, sess: CheckerSession):
"""
Check a local dev link; if it has an anchor, use the Chrome driver to
check for the presence of an element with a matching ID, otherwise fall
@@ -373,32 +427,29 @@ class LinkChecker:
return (1, self.anchor_cache[href])
id = href.split("#",1)[1].strip()
# Use the selenium driver to check for the exact anchor
self.chrome2.get(href)
sleep(0.3) # delay to give it time for hydration failures
try:
el = self.chrome2.find_element(By.ID, id)
except NoSuchElementException:
el = None
if el:
self.anchor_cache[href] = True
if sess.find_id_in_page(href, id):
with self.cache_lock:
self.anchor_cache[href] = True
return (1, True)
else:
self.anchor_cache[href] = False
with self.cache_lock:
self.anchor_cache[href] = False
return (1, False)
return self.check_link(href)
return self.check_link(href, sess)
def report(self, broken_links: list, total_links_checked: int):
def report(self):
print("---------------------------------------------------------------")
print(f"{len(broken_links)} broken links found among {total_links_checked} total links.")
print(f"{len(self.broken_links)} broken links found among "
f"{self.total_links_checked} total links.")
last_printed_in_file = None
for in_file, href in broken_links:
for in_file, href in self.broken_links:
if in_file != last_printed_in_file:
print("File:", in_file)
last_printed_in_file = in_file
print(" Link:", href)
if broken_links:
self.write_broken_links_report(broken_links, total_links_checked)
if self.broken_links:
self.write_broken_links_report()
else:
# Clean up any stale report from a previous run so the file's
# presence is a reliable signal for "broken links exist."
@@ -409,22 +460,22 @@ class LinkChecker:
os.remove(report_path)
logger.info(f"Removed stale {report_path} (no broken links this run).")
def write_broken_links_report(self, broken_links: list, total_links_checked: int):
def write_broken_links_report(self):
"""
Write a Markdown report of broken links to BROKEN_LINKS_REPORT_FILE,
grouped by the file that contained each link. Only called when at
least one broken link was found.
"""
by_file = {}
for in_file, href in broken_links:
for in_file, href in self.broken_links:
by_file.setdefault(in_file, []).append(href)
lines = []
lines.append("# Broken links report")
lines.append("")
lines.append(
f"**{len(broken_links)} broken link(s) found** in {len(by_file)} file(s) "
f"(out of {total_links_checked} total links checked)."
f"**{len(self.broken_links)} broken link(s) found** in {len(by_file)} file(s) "
f"(out of {self.total_links_checked} total links checked)."
)
lines.append("")
@@ -450,8 +501,6 @@ if __name__ == "__main__":
help="Suppress informational status messages")
noisiness.add_argument("--debug", "-d", action="store_true",
help="Print debug-level log messages")
parser.add_argument("--live", "-l", action="store_true",
help="Use a Selenium-powered browser session to get rendered docs")
parser.add_argument("path", type=str, nargs="?", default="docs/",
help="Check *.md files in this directory (including subdirs)")
@@ -463,7 +512,7 @@ if __name__ == "__main__":
else:
logger.setLevel(logging.INFO)
l = LinkChecker(cli_args.path, live=cli_args.live)
l = LinkChecker(cli_args.path)
try:
l.walk()
except (KeyboardInterrupt) as e: