Public Access
Site / build (pull_request) Successful in 8s
"The last post" and "the first post" in the LoRa, Gemini and S1 posts are now links. check_site.py follows every link to another page of the site and the #fragment it names, so a broken one fails the Site job. Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01EhqxQ49eCju4CzKYNjZzwT
126 lines
5.2 KiB
Python
Executable File
126 lines
5.2 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Checks a built site (docs/milestones/W1.md): what the pages promise, kept.
|
|
|
|
Usage: site/tools/check_site.py <built site folder>
|
|
|
|
- No page loads anything from another origin: no script, stylesheet, image, font, frame or preload
|
|
from elsewhere (links a visitor follows are fine).
|
|
- Every page has a title, a description and a language, and every image has an alt attribute.
|
|
- Every file a page references on this site exists.
|
|
- Every link to another page of the site lands on a page that exists, and so does the #fragment it names.
|
|
"""
|
|
import html.parser
|
|
import os
|
|
import re
|
|
import sys
|
|
import urllib.parse
|
|
|
|
ROOT = sys.argv[1] if len(sys.argv) > 1 else "public"
|
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
|
BASE = re.search(r'base_url\s*=\s*"([^"]+)"', open(os.path.join(HERE, "..", "config.toml")).read()).group(1)
|
|
HOST = urllib.parse.urlparse(BASE).netloc
|
|
|
|
|
|
class Page(html.parser.HTMLParser):
|
|
def __init__(self):
|
|
super().__init__()
|
|
self.title = False
|
|
self.has_title = False
|
|
self.has_description = False
|
|
self.lang = None
|
|
self.loads = [] # (tag, url) of everything the page fetches by itself
|
|
self.images_without_alt = 0
|
|
self.links = [] # href of every <a>
|
|
self.ids = set() # every id and name: what a #fragment can name
|
|
|
|
def handle_starttag(self, tag, attrs):
|
|
a = dict(attrs)
|
|
for key in ("id", "name"):
|
|
if a.get(key):
|
|
self.ids.add(a[key])
|
|
if tag == "a" and a.get("href"):
|
|
self.links.append(a["href"])
|
|
if tag == "html":
|
|
self.lang = a.get("lang")
|
|
if tag == "title":
|
|
self.title = True
|
|
if tag == "meta" and a.get("name") == "description" and a.get("content"):
|
|
self.has_description = True
|
|
if tag in ("script", "img", "iframe", "frame", "embed", "source", "video", "audio") and a.get("src"):
|
|
self.loads.append((tag, a["src"]))
|
|
if tag == "link" and a.get("href") and a.get("rel", "") in ("stylesheet", "preload", "icon", "modulepreload", "prefetch"):
|
|
self.loads.append(("link " + a["rel"], a["href"]))
|
|
if tag == "img" and "alt" not in a:
|
|
self.images_without_alt += 1
|
|
|
|
def handle_data(self, data):
|
|
if self.title and data.strip():
|
|
self.has_title = True
|
|
|
|
def handle_endtag(self, tag):
|
|
if tag == "title":
|
|
self.title = False
|
|
|
|
|
|
problems = []
|
|
pages = 0
|
|
parsed = {} # path of a page, from the site's root -> its Page
|
|
for folder, _, files in os.walk(ROOT):
|
|
for name in files:
|
|
if not name.endswith(".html"):
|
|
continue
|
|
path = os.path.join(folder, name)
|
|
shown = os.path.relpath(path, ROOT)
|
|
p = Page()
|
|
p.feed(open(path, encoding="utf-8").read())
|
|
pages += 1
|
|
parsed["/" + shown.replace(os.sep, "/")] = (p, folder)
|
|
if not p.has_title: problems.append(f"{shown}: no title")
|
|
if not p.has_description: problems.append(f"{shown}: no description")
|
|
if not p.lang: problems.append(f"{shown}: no language")
|
|
if p.images_without_alt: problems.append(f"{shown}: {p.images_without_alt} image(s) without alt")
|
|
for tag, url in p.loads:
|
|
u = urllib.parse.urlparse(url)
|
|
if u.netloc and u.netloc != HOST:
|
|
problems.append(f"{shown}: {tag} loads {url} from another origin")
|
|
elif not u.netloc and not u.scheme:
|
|
target = os.path.join(ROOT, u.path.lstrip("/")) if u.path.startswith("/") else os.path.join(folder, u.path)
|
|
if u.path and not os.path.exists(target):
|
|
problems.append(f"{shown}: {tag} {url} does not exist")
|
|
elif u.netloc == HOST:
|
|
target = os.path.join(ROOT, u.path.lstrip("/"))
|
|
if not os.path.exists(target):
|
|
problems.append(f"{shown}: {tag} {url} does not exist")
|
|
|
|
def page_of(url_path):
|
|
"""The built page a path of the site names: /guide/ is /guide/index.html."""
|
|
if url_path.endswith("/"):
|
|
return url_path + "index.html"
|
|
if os.path.isdir(os.path.join(ROOT, url_path.lstrip("/"))):
|
|
return url_path + "/index.html"
|
|
return url_path
|
|
|
|
|
|
for shown_path, (p, folder) in parsed.items():
|
|
for href in p.links:
|
|
u = urllib.parse.urlparse(href)
|
|
if u.scheme in ("mailto", "tel") or (u.netloc and u.netloc != HOST):
|
|
continue
|
|
if not u.path: # "#fragment": this page
|
|
target = shown_path
|
|
elif u.path.startswith("/"):
|
|
target = page_of(urllib.parse.unquote(u.path))
|
|
else:
|
|
target = page_of(os.path.normpath(os.path.join(os.path.dirname(shown_path), urllib.parse.unquote(u.path))).replace(os.sep, "/"))
|
|
if not os.path.exists(os.path.join(ROOT, target.lstrip("/"))):
|
|
problems.append(f"{shown_path.lstrip('/')}: link {href} goes nowhere")
|
|
elif u.fragment and target in parsed and u.fragment not in parsed[target][0].ids:
|
|
problems.append(f"{shown_path.lstrip('/')}: link {href}: no #{u.fragment} on that page")
|
|
|
|
if not pages:
|
|
problems.append(f"no pages found in {ROOT}")
|
|
for p in problems:
|
|
print("site check:", p)
|
|
print(f"site check: {pages} pages, {len(problems)} problem(s)")
|
|
sys.exit(1 if problems else 0)
|