Files
roro9stack/site/tools/check_site.py
T
twislaandClaude Sonnet 5.5 8f4b5e0ddc Site: roro9stack.net, phase 1 (the home page, Install with a browser flasher, Downloads)
A Zola site in site/, from the design: both themes on the device's
palette, notched shapes, self-hosted fonts, the wordmark and icons, two
generated hero drawings, nine App cards (the mesh messenger marked
planned), real screenshots, the updates and build sections. The Install
page builds ESP Web Tools' manifest in the browser from the Gitea API (it
needs Caddy to allow the origin), refuses any download that isn't on the
project's server, and falls back to the esptool steps. Downloads lists the
releases. The focus ring shows on notched controls. A page checker fails
the build if a page loads from another origin.

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01EhqxQ49eCju4CzKYNjZzwT
2026-10-06 18:20:18 +02:00

91 lines
3.5 KiB
Python
Executable File

#!/usr/bin/env python3
"""Checks a built site (docs/milestones/W1.md): what the pages promise, kept.
Usage: site/tools/check_site.py <built site folder>
- No page loads anything from another origin: no script, stylesheet, image, font, frame or preload
from elsewhere (links a visitor follows are fine).
- Every page has a title, a description and a language, and every image has an alt attribute.
- Every file a page references on this site exists.
"""
import html.parser
import os
import re
import sys
import urllib.parse
ROOT = sys.argv[1] if len(sys.argv) > 1 else "public"
HERE = os.path.dirname(os.path.abspath(__file__))
BASE = re.search(r'base_url\s*=\s*"([^"]+)"', open(os.path.join(HERE, "..", "config.toml")).read()).group(1)
HOST = urllib.parse.urlparse(BASE).netloc
class Page(html.parser.HTMLParser):
def __init__(self):
super().__init__()
self.title = False
self.has_title = False
self.has_description = False
self.lang = None
self.loads = [] # (tag, url) of everything the page fetches by itself
self.images_without_alt = 0
def handle_starttag(self, tag, attrs):
a = dict(attrs)
if tag == "html":
self.lang = a.get("lang")
if tag == "title":
self.title = True
if tag == "meta" and a.get("name") == "description" and a.get("content"):
self.has_description = True
if tag in ("script", "img", "iframe", "frame", "embed", "source", "video", "audio") and a.get("src"):
self.loads.append((tag, a["src"]))
if tag == "link" and a.get("href") and a.get("rel", "") in ("stylesheet", "preload", "icon", "modulepreload", "prefetch"):
self.loads.append(("link " + a["rel"], a["href"]))
if tag == "img" and "alt" not in a:
self.images_without_alt += 1
def handle_data(self, data):
if self.title and data.strip():
self.has_title = True
def handle_endtag(self, tag):
if tag == "title":
self.title = False
problems = []
pages = 0
for folder, _, files in os.walk(ROOT):
for name in files:
if not name.endswith(".html"):
continue
path = os.path.join(folder, name)
shown = os.path.relpath(path, ROOT)
p = Page()
p.feed(open(path, encoding="utf-8").read())
pages += 1
if not p.has_title: problems.append(f"{shown}: no title")
if not p.has_description: problems.append(f"{shown}: no description")
if not p.lang: problems.append(f"{shown}: no language")
if p.images_without_alt: problems.append(f"{shown}: {p.images_without_alt} image(s) without alt")
for tag, url in p.loads:
u = urllib.parse.urlparse(url)
if u.netloc and u.netloc != HOST:
problems.append(f"{shown}: {tag} loads {url} from another origin")
elif not u.netloc and not u.scheme:
target = os.path.join(ROOT, u.path.lstrip("/")) if u.path.startswith("/") else os.path.join(folder, u.path)
if u.path and not os.path.exists(target):
problems.append(f"{shown}: {tag} {url} does not exist")
elif u.netloc == HOST:
target = os.path.join(ROOT, u.path.lstrip("/"))
if not os.path.exists(target):
problems.append(f"{shown}: {tag} {url} does not exist")
if not pages:
problems.append(f"no pages found in {ROOT}")
for p in problems:
print("site check:", p)
print(f"site check: {pages} pages, {len(problems)} problem(s)")
sys.exit(1 if problems else 0)