#!/usr/bin/env python3 """Checks a built site (docs/milestones/W1.md): what the pages promise, kept. Usage: site/tools/check_site.py - No page loads anything from another origin: no script, stylesheet, image, font, frame or preload from elsewhere (links a visitor follows are fine). - Every page has a title, a description and a language, and every image has an alt attribute. - Every file a page references on this site exists. - Every link to another page of the site lands on a page that exists, and so does the #fragment it names. """ import html.parser import os import re import sys import urllib.parse ROOT = sys.argv[1] if len(sys.argv) > 1 else "public" HERE = os.path.dirname(os.path.abspath(__file__)) BASE = re.search(r'base_url\s*=\s*"([^"]+)"', open(os.path.join(HERE, "..", "config.toml")).read()).group(1) HOST = urllib.parse.urlparse(BASE).netloc class Page(html.parser.HTMLParser): def __init__(self): super().__init__() self.title = False self.has_title = False self.has_description = False self.lang = None self.loads = [] # (tag, url) of everything the page fetches by itself self.images_without_alt = 0 self.links = [] # href of every self.ids = set() # every id and name: what a #fragment can name def handle_starttag(self, tag, attrs): a = dict(attrs) for key in ("id", "name"): if a.get(key): self.ids.add(a[key]) if tag == "a" and a.get("href"): self.links.append(a["href"]) if tag == "html": self.lang = a.get("lang") if tag == "title": self.title = True if tag == "meta" and a.get("name") == "description" and a.get("content"): self.has_description = True if tag in ("script", "img", "iframe", "frame", "embed", "source", "video", "audio") and a.get("src"): self.loads.append((tag, a["src"])) if tag == "link" and a.get("href") and a.get("rel", "") in ("stylesheet", "preload", "icon", "modulepreload", "prefetch"): self.loads.append(("link " + a["rel"], a["href"])) if tag == "img" and "alt" not in a: self.images_without_alt += 1 def handle_data(self, data): if self.title and data.strip(): self.has_title = True def handle_endtag(self, tag): if tag == "title": self.title = False problems = [] pages = 0 parsed = {} # path of a page, from the site's root -> its Page for folder, _, files in os.walk(ROOT): for name in files: if not name.endswith(".html"): continue path = os.path.join(folder, name) shown = os.path.relpath(path, ROOT) p = Page() p.feed(open(path, encoding="utf-8").read()) pages += 1 parsed["/" + shown.replace(os.sep, "/")] = (p, folder) if not p.has_title: problems.append(f"{shown}: no title") if not p.has_description: problems.append(f"{shown}: no description") if not p.lang: problems.append(f"{shown}: no language") if p.images_without_alt: problems.append(f"{shown}: {p.images_without_alt} image(s) without alt") for tag, url in p.loads: u = urllib.parse.urlparse(url) if u.netloc and u.netloc != HOST: problems.append(f"{shown}: {tag} loads {url} from another origin") elif not u.netloc and not u.scheme: target = os.path.join(ROOT, u.path.lstrip("/")) if u.path.startswith("/") else os.path.join(folder, u.path) if u.path and not os.path.exists(target): problems.append(f"{shown}: {tag} {url} does not exist") elif u.netloc == HOST: target = os.path.join(ROOT, u.path.lstrip("/")) if not os.path.exists(target): problems.append(f"{shown}: {tag} {url} does not exist") def page_of(url_path): """The built page a path of the site names: /guide/ is /guide/index.html.""" if url_path.endswith("/"): return url_path + "index.html" if os.path.isdir(os.path.join(ROOT, url_path.lstrip("/"))): return url_path + "/index.html" return url_path for shown_path, (p, folder) in parsed.items(): for href in p.links: u = urllib.parse.urlparse(href) if u.scheme in ("mailto", "tel") or (u.netloc and u.netloc != HOST): continue if not u.path: # "#fragment": this page target = shown_path elif u.path.startswith("/"): target = page_of(urllib.parse.unquote(u.path)) else: target = page_of(os.path.normpath(os.path.join(os.path.dirname(shown_path), urllib.parse.unquote(u.path))).replace(os.sep, "/")) if not os.path.exists(os.path.join(ROOT, target.lstrip("/"))): problems.append(f"{shown_path.lstrip('/')}: link {href} goes nowhere") elif u.fragment and target in parsed and u.fragment not in parsed[target][0].ids: problems.append(f"{shown_path.lstrip('/')}: link {href}: no #{u.fragment} on that page") if not pages: problems.append(f"no pages found in {ROOT}") for p in problems: print("site check:", p) print(f"site check: {pages} pages, {len(problems)} problem(s)") sys.exit(1 if problems else 0)