mirror of
https://github.com/XRPLF/xrpl-dev-portal.git
synced 2026-07-28 01:20:20 +00:00
Parallelize & clean up link checker
This commit is contained in:
2
.github/workflows/selenium-link-check.yml
vendored
2
.github/workflows/selenium-link-check.yml
vendored
@@ -62,7 +62,7 @@ jobs:
|
||||
- name: Run Selenium link checker
|
||||
run: |
|
||||
set -eo pipefail
|
||||
python tools/check-external-links.py --live docs/ 2>&1 | tee selenium-link-check.log
|
||||
python tools/check-links.py docs/ 2>&1 | tee selenium-link-check.log
|
||||
|
||||
- name: Stop Realm server
|
||||
if: always()
|
||||
|
||||
@@ -12,12 +12,14 @@
|
||||
# * may be used as a wildcard AT THE END OF LINKS ONLY
|
||||
|
||||
# These sites block crawlers (often using Cloudflare) with a 403 but typically
|
||||
# work when accessed from an actual browser:
|
||||
# work when accessed from an actual browser. It's a good idea to re-check them
|
||||
# manually every once in a while to make sure they really do still work.
|
||||
https://x.com/*
|
||||
https://www.gnu.org/software/screen/manual/screen.html
|
||||
https://www.coindesk.com/markets/2015/04/02/1-million-legal-fight-ensnares-ripple-bitstamp-and-jed-mccaleb/
|
||||
https://www.coindesk.com/markets/2016/02/12/ripple-settles-1-million-lawsuit-with-former-executive-and-founder/
|
||||
https://bsaaml.ffiec.gov/manual/Introduction/01
|
||||
https://bsaaml.ffiec.gov/manual/AssessingComplianceWithBSARegulatoryRequirements/04
|
||||
https://www.sec.gov/oiea/investor-alerts-and-bulletins/ib_coinofferings
|
||||
https://www.npmjs.com/*
|
||||
https://go.dev/dl/
|
||||
|
||||
@@ -7,13 +7,11 @@ tools/check-external-links.py [folder/to/check/]
|
||||
|
||||
If [folder/to/check] is omitted, check ./docs/
|
||||
|
||||
Prints a report of broken links. In the default mode, checks external links in
|
||||
Markdown files only. Pass --live to test against a local Redocly dev server,
|
||||
which checks *all* links, including anchors and markdoc tags; assumes the
|
||||
Prints a report of broken links. Tests against a local Redocly dev server,
|
||||
and checks *all* links, including anchors and markdoc tags; assumes the
|
||||
server is already up and running on localhost:4000.
|
||||
|
||||
Requires: beautifulsoup4, markdown, requests
|
||||
For live mode, selenium and Chrome are also required.
|
||||
Requires: requests, selenium (w/ Chrome WebDriver)
|
||||
|
||||
Links & sites that often report false-positives can be added to broken-links.txt
|
||||
to have the link checker skip them.
|
||||
@@ -24,12 +22,11 @@ import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import threading
|
||||
from time import time, sleep
|
||||
|
||||
import requests
|
||||
from requests.adapters import HTTPAdapter, Retry
|
||||
from markdown import markdown
|
||||
from bs4 import BeautifulSoup
|
||||
from selenium import webdriver
|
||||
from selenium.webdriver.common.by import By
|
||||
from selenium.common.exceptions import StaleElementReferenceException, NoSuchElementException
|
||||
@@ -45,6 +42,8 @@ DEFAULT_SKIP_PATHS = [
|
||||
"_code-samples", # Debatably, we might want to link-check the READMEs here
|
||||
"_api-examples",
|
||||
"_sources",
|
||||
"_snippets",
|
||||
"img",
|
||||
]
|
||||
MAX_RETRIES = 1 # Times to retry if a link doesn't work
|
||||
TIMEOUT_SECONDS = 8 # Seconds before giving up on a link
|
||||
@@ -63,23 +62,110 @@ logger = logging.getLogger(__name__)
|
||||
logger.addHandler(logging.StreamHandler())
|
||||
logger.propagate = False
|
||||
|
||||
class CheckerSession:
|
||||
"""
|
||||
Instance of all the link-checking tools to be used by a given thread,
|
||||
including two Chrome webdrivers and two Requests sessions with automatic
|
||||
backoff and custom user-agent.
|
||||
"""
|
||||
def __init__(self):
|
||||
retries = Retry(total=MAX_RETRIES, backoff_factor=1, status_forcelist=[ 502, 503, 504 ])
|
||||
self.s = requests.Session()
|
||||
self.s.mount("https://", HTTPAdapter(max_retries=retries))
|
||||
self.h = requests.Session()
|
||||
self.h.mount("http://", HTTPAdapter(max_retries=retries))
|
||||
self.h.headers.update({"User-Agent": USER_AGENT})
|
||||
self.s.headers.update({"User-Agent": USER_AGENT})
|
||||
self.last_host_called = None
|
||||
|
||||
options = webdriver.ChromeOptions()
|
||||
options.add_argument("--headless=new")
|
||||
self.chrome = webdriver.Chrome(options=options)
|
||||
# Make a second one so we can check for anchors without resetting
|
||||
# all the references we have in the page the link comes from
|
||||
self.chrome2 = webdriver.Chrome(options=options)
|
||||
|
||||
def fetch_code(self, href: str):
|
||||
"""
|
||||
Get status code of a URL, using saved sessions & retries, automatically
|
||||
failing over from HTTP HEAD to HTTP GET and adding a slight delay to
|
||||
avoid hammering the same host repeatedly.
|
||||
"""
|
||||
proto,predicate = href.split("//",1)
|
||||
host = predicate.split("/", 1)[0]
|
||||
if host == self.last_host_called:
|
||||
sleep(SAME_HOST_DELAY)
|
||||
self.last_host_called = host
|
||||
if href.startswith("http://"):
|
||||
sess = self.h
|
||||
else:
|
||||
sess = self.s
|
||||
try:
|
||||
code = sess.head(href, timeout=TIMEOUT_SECONDS).status_code
|
||||
except Exception as e:
|
||||
logger.debug(f"Error getting {href}: {e}")
|
||||
try:
|
||||
code = sess.get(href, timeout=TIMEOUT_SECONDS).status_code
|
||||
except Exception as e2:
|
||||
logger.debug(f"Error getting {href}: {e2}")
|
||||
code = 500
|
||||
return code
|
||||
|
||||
def get_page_hrefs_and_text(self, href: str):
|
||||
self.chrome.get(href)
|
||||
rootlayout = self.chrome.find_element(By.CSS_SELECTOR, '[data-component-name="layouts/RootLayout"]')
|
||||
try:
|
||||
links = rootlayout.find_elements(By.CSS_SELECTOR, "a")
|
||||
pagetext = rootlayout.text
|
||||
hrefs = [link.get_attribute("href") for link in links]
|
||||
except StaleElementReferenceException:
|
||||
# This can happen when hydration fails or the page is updated
|
||||
# asynchronously, for example by the amendment-disclaimer tag.
|
||||
# Try again and hopefully it works this time.
|
||||
sleep(0.2)
|
||||
rootlayout = self.chrome.find_element(By.CSS_SELECTOR, '[data-component-name="layouts/RootLayout"]')
|
||||
pagetext = rootlayout.text
|
||||
links = rootlayout.find_elements(By.CSS_SELECTOR, "a")
|
||||
hrefs = [link.get_attribute("href") for link in links]
|
||||
return hrefs, pagetext
|
||||
|
||||
def find_id_in_page(self, href: str, id: str):
|
||||
"""
|
||||
Return the element with a given unique ID in the DOM, or None if no
|
||||
matching element is found. Uses the second chrome driver to avoid
|
||||
messing with the first driver's view of the page the link is from
|
||||
"""
|
||||
self.chrome2.get(href)
|
||||
sleep(0.3) # delay to give it time for hydration failures
|
||||
try:
|
||||
el = self.chrome2.find_element(By.ID, id)
|
||||
except NoSuchElementException:
|
||||
el = None
|
||||
return el
|
||||
|
||||
class LinkChecker:
|
||||
def __init__(self, topdir,
|
||||
skip_paths = DEFAULT_SKIP_PATHS,
|
||||
cache_file = DEFAULT_CACHE_FILE,
|
||||
known_broken = KNOWN_BROKEN_LINKS_FILE,
|
||||
live = False
|
||||
known_broken = KNOWN_BROKEN_LINKS_FILE
|
||||
):
|
||||
self.topdir = topdir
|
||||
self.skip_paths = skip_paths
|
||||
self.last_checkin = time()
|
||||
self.last_cache_update = 0
|
||||
self.live = live
|
||||
self.setup_sessions()
|
||||
self.init_cache(cache_file)
|
||||
self.init_known_broken(known_broken)
|
||||
|
||||
def init_cache(self, cache_file: str):
|
||||
"""
|
||||
Set up a local cache of link checker results so multiple links to the
|
||||
same place don't have to be checked repeatedly. The cache can be saved
|
||||
to a JSON file and loaded from it to save time across runs, but even if
|
||||
it isn't, a cache dict is used during a single run to save time.
|
||||
The cache uses a lock so that multiple threads don't write to it at the
|
||||
same time, but reading from the cache doesn't require the lock.
|
||||
"""
|
||||
self.cache_lock = threading.Lock()
|
||||
self.anchor_cache = {}
|
||||
if not cache_file:
|
||||
logger.debug("No cache file, not loading anything.")
|
||||
@@ -116,31 +202,23 @@ class LinkChecker:
|
||||
self.last_cache_update = time()
|
||||
|
||||
def write_cache(self):
|
||||
if not self.cache_file:
|
||||
return
|
||||
with open(self.cache_file, "w") as f:
|
||||
json.dump(self.cache, f)
|
||||
self.last_cache_update = time()
|
||||
|
||||
def setup_sessions(self):
|
||||
retries = Retry(total=MAX_RETRIES, backoff_factor=1, status_forcelist=[ 502, 503, 504 ])
|
||||
self.s = requests.Session()
|
||||
self.s.mount("https://", HTTPAdapter(max_retries=retries))
|
||||
self.h = requests.Session()
|
||||
self.h.mount("http://", HTTPAdapter(max_retries=retries))
|
||||
self.h.headers.update({"User-Agent": USER_AGENT})
|
||||
self.s.headers.update({"User-Agent": USER_AGENT})
|
||||
self.last_host_called = None
|
||||
|
||||
if self.live:
|
||||
options = webdriver.ChromeOptions()
|
||||
options.add_argument("--headless=new")
|
||||
self.chrome = webdriver.Chrome(options=options)
|
||||
# Make a second one so we can check for anchors without resetting
|
||||
# all the references we have in the page the link comes from
|
||||
self.chrome2 = webdriver.Chrome(options=options)
|
||||
"""
|
||||
Update the cache file with the latest of the cache, so even if the run
|
||||
doesn't finish successfully, some progress is saved.
|
||||
"""
|
||||
with self.cache_lock:
|
||||
if not self.cache_file:
|
||||
return
|
||||
with open(self.cache_file, "w") as f:
|
||||
json.dump(self.cache, f)
|
||||
self.last_cache_update = time()
|
||||
|
||||
def init_known_broken(self, known_broken):
|
||||
"""
|
||||
Read the known broken links config file and set up internal structures
|
||||
for recognizing 'known broken' links that should be excluded from
|
||||
checking, based on its contents.
|
||||
"""
|
||||
self.exact_known_broken = []
|
||||
self.wildcard_known_broken = []
|
||||
|
||||
@@ -174,12 +252,51 @@ class LinkChecker:
|
||||
self.write_cache()
|
||||
|
||||
def walk(self):
|
||||
logger.info(f"Checking files in {os.path.abspath(self.topdir)}")
|
||||
externalCache = []
|
||||
"""
|
||||
Walk all folders in the top dir and check for
|
||||
broken links, paralellizing as possible.
|
||||
"""
|
||||
threads = []
|
||||
self.broken_links = []
|
||||
self.total_links_checked = 0
|
||||
self.report_lock = threading.Lock()
|
||||
dirnames = [d.name for d in os.scandir(self.topdir) if d.is_dir()]
|
||||
dirnames[:] = [d for d in dirnames if d not in self.skip_paths]
|
||||
for dirpath in dirnames:
|
||||
dirpath = os.path.join(self.topdir, dirpath)
|
||||
t = threading.Thread(target=self.check_dir, args=(dirpath,))
|
||||
threads.append(t)
|
||||
top_dir_thread = threading.Thread(target=self.check_dir,
|
||||
args=(self.topdir, False))
|
||||
threads.append(top_dir_thread)
|
||||
logger.info(f"Checking dirs: {'\n '.join(dirnames)}")
|
||||
logger.info(f"Starting {len(threads)} threads to check links...")
|
||||
for t in threads:
|
||||
t.start()
|
||||
for t in threads:
|
||||
t.join()
|
||||
self.report()
|
||||
|
||||
def record_broken(self, broken_links: list, num_checked: int):
|
||||
"""
|
||||
Helper for checker threads to report to the main thread with their
|
||||
lists of broken links
|
||||
"""
|
||||
with self.report_lock:
|
||||
self.broken_links += broken_links
|
||||
self.total_links_checked += num_checked
|
||||
|
||||
def check_dir(self, top_dirpath, recurse=True):
|
||||
"""
|
||||
Check a single directory (& subdirs) for broken links. Capable of
|
||||
running in a separate thread.
|
||||
"""
|
||||
logger.info(f"Checking files in {os.path.abspath(top_dirpath)}")
|
||||
broken_links = []
|
||||
total_links_checked = 0
|
||||
last_checkin = time()
|
||||
for dirpath, dirnames, filenames in os.walk(self.topdir):
|
||||
sess = CheckerSession()
|
||||
for dirpath, dirnames, filenames in os.walk(top_dirpath):
|
||||
dirnames[:] = [d for d in dirnames if d not in self.skip_paths]
|
||||
self.checkin(f"dir: {dirpath}")
|
||||
if dirpath in self.skip_paths:
|
||||
@@ -188,51 +305,17 @@ class LinkChecker:
|
||||
for fname in filenames:
|
||||
self.checkin(f"file: {fname}")
|
||||
in_file = os.path.join(dirpath, fname)
|
||||
|
||||
if in_file.endswith(".md"):
|
||||
if self.live:
|
||||
newly_checked, newly_broken = self.check_file_live(in_file)
|
||||
else:
|
||||
newly_checked, newly_broken = self.check_file(in_file)
|
||||
if in_file.endswith(".md") or in_file.endswith(".page.tsx"):
|
||||
newly_checked, newly_broken = self.check_file(in_file, sess)
|
||||
broken_links += newly_broken
|
||||
total_links_checked += newly_checked
|
||||
if in_file.endswith(".page.tsx") and self.live:
|
||||
newly_checked, newly_broken = self.check_file_live(in_file)
|
||||
broken_links += newly_broken
|
||||
total_links_checked += newly_checked
|
||||
self.report(broken_links, total_links_checked)
|
||||
else:
|
||||
logger.debug(f"Not checking non-page file {in_file}")
|
||||
if not recurse:
|
||||
break
|
||||
self.record_broken(broken_links, total_links_checked)
|
||||
|
||||
def check_file(self, in_file: str):
|
||||
"""
|
||||
Given a specific .md file, look for external links in it and check them.
|
||||
Returns how many were checked and a list of tuples where each member is
|
||||
a (file,link) pair representing a broken link.
|
||||
|
||||
Note that this does not parse Markdoc including partials so it can't
|
||||
handle the full context of the Redocly parser but those *usually* won't
|
||||
be needed for external links.
|
||||
"""
|
||||
logger.info(f"Checking file {in_file}")
|
||||
with open(in_file, 'r', encoding="utf-8") as f:
|
||||
html = markdown(f.read())
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
links = soup.find_all("a")
|
||||
broken = []
|
||||
num_checked = 0
|
||||
for link in links:
|
||||
self.checkin(f"link: {link}")
|
||||
if "href" not in link.attrs:
|
||||
# probably an <a name> type anchor, skip
|
||||
continue
|
||||
if (link["href"].startswith("https://") or
|
||||
link["href"].startswith("http://")):
|
||||
was_checked, was_good = self.check_link(link["href"])
|
||||
num_checked += was_checked
|
||||
if not was_good:
|
||||
broken.append( (in_file, link["href"]) )
|
||||
return num_checked, broken
|
||||
|
||||
def check_file_live(self, in_file: str):
|
||||
def check_file(self, in_file: str, sess: CheckerSession):
|
||||
"""
|
||||
Given a specific .md file, fetch it from the Redocly dev server and
|
||||
check for links in it.
|
||||
@@ -246,7 +329,7 @@ class LinkChecker:
|
||||
logger.warning(f"Not checking path that's not an md or page.tsx file: {in_file}")
|
||||
return (0, [])
|
||||
url = REDOCLY_DEV_BASE + path
|
||||
code = self.fetch(url)
|
||||
code = sess.fetch_code(url)
|
||||
if code < 200 or code >= 400:
|
||||
logger.warning(f"Failed to get page from dev server for file {in_file}")
|
||||
return 0, []
|
||||
@@ -254,34 +337,19 @@ class LinkChecker:
|
||||
broken = []
|
||||
num_checked = 0
|
||||
logger.info(f"Checking path {path}")
|
||||
self.chrome.get(REDOCLY_DEV_BASE+path)
|
||||
# sleep(0.2) # give it a moment in case hydration fails or something
|
||||
rootlayout = self.chrome.find_element(By.CSS_SELECTOR, '[data-component-name="layouts/RootLayout"]')
|
||||
try:
|
||||
links = rootlayout.find_elements(By.CSS_SELECTOR, "a")
|
||||
pagetext = rootlayout.text
|
||||
hrefs = [link.get_attribute("href") for link in links]
|
||||
except StaleElementReferenceException:
|
||||
# This can happen when hydration fails or the page is updated
|
||||
# asynchronously, for example by fetching amendment status.
|
||||
# Try again and hopefully it works this time.
|
||||
sleep(0.2)
|
||||
rootlayout = self.chrome.find_element(By.CSS_SELECTOR, '[data-component-name="layouts/RootLayout"]')
|
||||
pagetext = rootlayout.text
|
||||
links = rootlayout.find_elements(By.CSS_SELECTOR, "a")
|
||||
hrefs = [link.get_attribute("href") for link in links]
|
||||
hrefs, pagetext = sess.get_page_hrefs_and_text(REDOCLY_DEV_BASE+path)
|
||||
for href in hrefs:
|
||||
self.checkin(f"link: {href}")
|
||||
if not href:
|
||||
# Probably a name anchor something, skip
|
||||
continue
|
||||
if href.startswith(REDOCLY_DEV_BASE):
|
||||
was_checked, was_good = self.check_dev_link(href)
|
||||
was_checked, was_good = self.check_dev_link(href, sess)
|
||||
num_checked += was_checked
|
||||
if not was_good:
|
||||
broken.append( (in_file, href) )
|
||||
elif href.startswith("http://") or href.startswith("https://"):
|
||||
was_checked, was_good = self.check_link(href)
|
||||
was_checked, was_good = self.check_link(href, sess)
|
||||
num_checked += was_checked
|
||||
if not was_good:
|
||||
broken.append( (in_file, href) )
|
||||
@@ -291,6 +359,10 @@ class LinkChecker:
|
||||
return num_checked, broken
|
||||
|
||||
def check_for_unparsed_reflinks(self, text: str):
|
||||
"""
|
||||
Given text of a page, find patterns like [Hash][] that should have been
|
||||
parsed into links from the Markdown source, but weren't.
|
||||
"""
|
||||
unparsed_links = []
|
||||
matches = UNMATCHED_REFLINK_REGEX.finditer(text)
|
||||
for m in matches:
|
||||
@@ -306,41 +378,22 @@ class LinkChecker:
|
||||
return True
|
||||
return False
|
||||
|
||||
def fetch(self, href: str):
|
||||
"""
|
||||
Get status code of a URL, using saved sessions & retries, automatically
|
||||
failing over from HTTP HEAD to HTTP GET and adding a slight delay to
|
||||
avoid hammering the same host repeatedly.
|
||||
"""
|
||||
proto,predicate = href.split("//",1)
|
||||
host = predicate.split("/", 1)[0]
|
||||
if host == self.last_host_called:
|
||||
sleep(SAME_HOST_DELAY)
|
||||
self.last_host_called = host
|
||||
|
||||
if href.startswith("http://"):
|
||||
sess = self.h
|
||||
else:
|
||||
sess = self.s
|
||||
|
||||
try:
|
||||
code = sess.head(href, timeout=TIMEOUT_SECONDS).status_code
|
||||
except Exception as e:
|
||||
logger.debug(f"Error getting {href}: {e}")
|
||||
try:
|
||||
code = sess.get(href, timeout=TIMEOUT_SECONDS).status_code
|
||||
except Exception as e2:
|
||||
logger.debug(f"Error getting {href}: {e2}")
|
||||
code = 500
|
||||
|
||||
return code
|
||||
|
||||
def trim_trackers(self, href: str):
|
||||
"""
|
||||
Remove query parameters that are added by JavaScript trackers. These
|
||||
parameters tend to include timestamps that bust the cache unnecessarily.
|
||||
"""
|
||||
if "?__hstc=" in href:
|
||||
return href[:href.find("?__hstc=")]
|
||||
return href
|
||||
|
||||
def check_link(self, href: str):
|
||||
def check_link(self, href: str, sess: CheckerSession):
|
||||
"""
|
||||
Check a link to see if it can be successfully fetched. Returns a tuple
|
||||
(num_checked: int, was_good: bool) where num_checked is 1 if the link
|
||||
was checked and 0 if it was skipped, and was_good is True if the link
|
||||
was fetched successfully and False if it failed.
|
||||
"""
|
||||
href = self.trim_trackers(href)
|
||||
if href in self.cache.keys():
|
||||
logger.debug(f"... Skipping (cached): {href}")
|
||||
@@ -350,16 +403,17 @@ class LinkChecker:
|
||||
logger.debug(f"... Skipping (known broken): {href}")
|
||||
return (0, True)
|
||||
logger.info(f"... Testing link {href}")
|
||||
code = self.fetch(href)
|
||||
code = sess.fetch_code(href)
|
||||
if code < 200 or code >= 400:
|
||||
logger.warning(f"... Broken link to {href}")
|
||||
self.cache[href] = (False, time())
|
||||
with self.cache_lock:
|
||||
self.cache[href] = (False, time())
|
||||
return (1, False)
|
||||
logger.info("... ... success.")
|
||||
self.cache[href] = (True, time())
|
||||
with self.cache_lock:
|
||||
self.cache[href] = (True, time())
|
||||
return (1, True)
|
||||
|
||||
def check_dev_link(self, href: str):
|
||||
def check_dev_link(self, href: str, sess: CheckerSession):
|
||||
"""
|
||||
Check a local dev link; if it has an anchor, use the Chrome driver to
|
||||
check for the presence of an element with a matching ID, otherwise fall
|
||||
@@ -373,32 +427,29 @@ class LinkChecker:
|
||||
return (1, self.anchor_cache[href])
|
||||
id = href.split("#",1)[1].strip()
|
||||
# Use the selenium driver to check for the exact anchor
|
||||
self.chrome2.get(href)
|
||||
sleep(0.3) # delay to give it time for hydration failures
|
||||
try:
|
||||
el = self.chrome2.find_element(By.ID, id)
|
||||
except NoSuchElementException:
|
||||
el = None
|
||||
if el:
|
||||
self.anchor_cache[href] = True
|
||||
if sess.find_id_in_page(href, id):
|
||||
with self.cache_lock:
|
||||
self.anchor_cache[href] = True
|
||||
return (1, True)
|
||||
else:
|
||||
self.anchor_cache[href] = False
|
||||
with self.cache_lock:
|
||||
self.anchor_cache[href] = False
|
||||
return (1, False)
|
||||
return self.check_link(href)
|
||||
return self.check_link(href, sess)
|
||||
|
||||
def report(self, broken_links: list, total_links_checked: int):
|
||||
def report(self):
|
||||
print("---------------------------------------------------------------")
|
||||
print(f"{len(broken_links)} broken links found among {total_links_checked} total links.")
|
||||
print(f"{len(self.broken_links)} broken links found among "
|
||||
f"{self.total_links_checked} total links.")
|
||||
last_printed_in_file = None
|
||||
for in_file, href in broken_links:
|
||||
for in_file, href in self.broken_links:
|
||||
if in_file != last_printed_in_file:
|
||||
print("File:", in_file)
|
||||
last_printed_in_file = in_file
|
||||
print(" Link:", href)
|
||||
|
||||
if broken_links:
|
||||
self.write_broken_links_report(broken_links, total_links_checked)
|
||||
if self.broken_links:
|
||||
self.write_broken_links_report()
|
||||
else:
|
||||
# Clean up any stale report from a previous run so the file's
|
||||
# presence is a reliable signal for "broken links exist."
|
||||
@@ -409,22 +460,22 @@ class LinkChecker:
|
||||
os.remove(report_path)
|
||||
logger.info(f"Removed stale {report_path} (no broken links this run).")
|
||||
|
||||
def write_broken_links_report(self, broken_links: list, total_links_checked: int):
|
||||
def write_broken_links_report(self):
|
||||
"""
|
||||
Write a Markdown report of broken links to BROKEN_LINKS_REPORT_FILE,
|
||||
grouped by the file that contained each link. Only called when at
|
||||
least one broken link was found.
|
||||
"""
|
||||
by_file = {}
|
||||
for in_file, href in broken_links:
|
||||
for in_file, href in self.broken_links:
|
||||
by_file.setdefault(in_file, []).append(href)
|
||||
|
||||
lines = []
|
||||
lines.append("# Broken links report")
|
||||
lines.append("")
|
||||
lines.append(
|
||||
f"**{len(broken_links)} broken link(s) found** in {len(by_file)} file(s) "
|
||||
f"(out of {total_links_checked} total links checked)."
|
||||
f"**{len(self.broken_links)} broken link(s) found** in {len(by_file)} file(s) "
|
||||
f"(out of {self.total_links_checked} total links checked)."
|
||||
)
|
||||
lines.append("")
|
||||
|
||||
@@ -450,8 +501,6 @@ if __name__ == "__main__":
|
||||
help="Suppress informational status messages")
|
||||
noisiness.add_argument("--debug", "-d", action="store_true",
|
||||
help="Print debug-level log messages")
|
||||
parser.add_argument("--live", "-l", action="store_true",
|
||||
help="Use a Selenium-powered browser session to get rendered docs")
|
||||
parser.add_argument("path", type=str, nargs="?", default="docs/",
|
||||
help="Check *.md files in this directory (including subdirs)")
|
||||
|
||||
@@ -463,7 +512,7 @@ if __name__ == "__main__":
|
||||
else:
|
||||
logger.setLevel(logging.INFO)
|
||||
|
||||
l = LinkChecker(cli_args.path, live=cli_args.live)
|
||||
l = LinkChecker(cli_args.path)
|
||||
try:
|
||||
l.walk()
|
||||
except (KeyboardInterrupt) as e:
|
||||
Reference in New Issue
Block a user