Site: the posts link to each other wherever they refer to one; check_site.py checks every link of the site
Site / build (pull_request) Successful in 8s

"The last post" and "the first post" in the LoRa, Gemini and S1 posts are
now links. check_site.py follows every link to another page of the site and
the #fragment it names, so a broken one fails the Site job.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01EhqxQ49eCju4CzKYNjZzwT
This commit is contained in:
2026-10-06 20:55:27 +02:00
co-authored by Claude Sonnet 5.5
parent 6e54658ba4
commit b5bb3ab7d1
5 changed files with 40 additions and 4 deletions
+35
View File
@@ -7,6 +7,7 @@ Usage: site/tools/check_site.py <built site folder>
from elsewhere (links a visitor follows are fine).
- Every page has a title, a description and a language, and every image has an alt attribute.
- Every file a page references on this site exists.
- Every link to another page of the site lands on a page that exists, and so does the #fragment it names.
"""
import html.parser
import os
@@ -29,9 +30,16 @@ class Page(html.parser.HTMLParser):
self.lang = None
self.loads = [] # (tag, url) of everything the page fetches by itself
self.images_without_alt = 0
self.links = [] # href of every <a>
self.ids = set() # every id and name: what a #fragment can name
def handle_starttag(self, tag, attrs):
a = dict(attrs)
for key in ("id", "name"):
if a.get(key):
self.ids.add(a[key])
if tag == "a" and a.get("href"):
self.links.append(a["href"])
if tag == "html":
self.lang = a.get("lang")
if tag == "title":
@@ -56,6 +64,7 @@ class Page(html.parser.HTMLParser):
problems = []
pages = 0
parsed = {} # path of a page, from the site's root -> its Page
for folder, _, files in os.walk(ROOT):
for name in files:
if not name.endswith(".html"):
@@ -65,6 +74,7 @@ for folder, _, files in os.walk(ROOT):
p = Page()
p.feed(open(path, encoding="utf-8").read())
pages += 1
parsed["/" + shown.replace(os.sep, "/")] = (p, folder)
if not p.has_title: problems.append(f"{shown}: no title")
if not p.has_description: problems.append(f"{shown}: no description")
if not p.lang: problems.append(f"{shown}: no language")
@@ -82,6 +92,31 @@ for folder, _, files in os.walk(ROOT):
if not os.path.exists(target):
problems.append(f"{shown}: {tag} {url} does not exist")
def page_of(url_path):
"""The built page a path of the site names: /guide/ is /guide/index.html."""
if url_path.endswith("/"):
return url_path + "index.html"
if os.path.isdir(os.path.join(ROOT, url_path.lstrip("/"))):
return url_path + "/index.html"
return url_path
for shown_path, (p, folder) in parsed.items():
for href in p.links:
u = urllib.parse.urlparse(href)
if u.scheme in ("mailto", "tel") or (u.netloc and u.netloc != HOST):
continue
if not u.path: # "#fragment": this page
target = shown_path
elif u.path.startswith("/"):
target = page_of(urllib.parse.unquote(u.path))
else:
target = page_of(os.path.normpath(os.path.join(os.path.dirname(shown_path), urllib.parse.unquote(u.path))).replace(os.sep, "/"))
if not os.path.exists(os.path.join(ROOT, target.lstrip("/"))):
problems.append(f"{shown_path.lstrip('/')}: link {href} goes nowhere")
elif u.fragment and target in parsed and u.fragment not in parsed[target][0].ids:
problems.append(f"{shown_path.lstrip('/')}: link {href}: no #{u.fragment} on that page")
if not pages:
problems.append(f"no pages found in {ROOT}")
for p in problems: