checkhtml.py
Strict HTML parse, dead links, feed and sitemap
156 lines. This is the file the build actually runs, copied verbatim at build time.
- 1
"""Parse every generated page with html5lib in strict mode and report problems. - 2
- 3
Also does a few checks a parser won't: internal links resolve to real files, - 4
referenced assets exist, and no page accidentally references a third-party host. - 5
""" - 6
- 7
import os - 8
import re - 9
import sys - 10
- 11
HERE = os.path.dirname(os.path.abspath(__file__)) - 12
sys.path.insert(0, os.path.join(HERE, "pylib")) - 13
ROOT = os.path.abspath(os.path.join(HERE, "..")) - 14
- 15
import html5lib # noqa: E402 - 16
- 17
PAGES = ["index.html", "tokens/index.html", "vocabulary/index.html", - 18
"predict/index.html", "cost/index.html", "learn/index.html", "attention/index.html", - 19
"about/index.html", "changes/index.html", "checking/index.html"] - 20
- 21
# Hosts a page is allowed to link to in href/src. Anything else is a surprise. - 22
ALLOWED_EXTERNAL = {"github.com"} - 23
- 24
SITE = "https://sweedworks.com" - 25
- 26
- 27
def local_target(url): - 28
"""Map a site-absolute URL to a file on disk, or None if not local. - 29
- 30
Absolute URLs to our own domain count as local: og:image, og:url and - 31
canonical all have to be absolute, and they still need to resolve. - 32
""" - 33
if url.startswith(SITE): - 34
url = url[len(SITE):] or "/" - 35
if url.startswith(("http://", "https://", "mailto:", "#", "data:")): - 36
return None - 37
path = url.split("#")[0].split("?")[0] - 38
if not path.startswith("/"): - 39
return None - 40
fs = os.path.join(ROOT, path.lstrip("/")) - 41
if path.endswith("/"): - 42
fs = os.path.join(fs, "index.html") - 43
return fs - 44
- 45
- 46
def main(): - 47
problems = [] - 48
- 49
for page in PAGES: - 50
full = os.path.join(ROOT, page) - 51
src = open(full, encoding="utf-8").read() - 52
- 53
parser = html5lib.HTMLParser(strict=True) - 54
try: - 55
parser.parse(src) - 56
print(f"PASS {page:<20} parses clean") - 57
except Exception as e: - 58
problems.append(f"{page}: parse error: {e}") - 59
print(f"FAIL {page:<20} {str(e)[:120]}") - 60
continue - 61
- 62
# Links and assets - 63
for attr in ("href", "src", "content"): - 64
for url in re.findall(rf'{attr}="([^"]+)"', src): - 65
if url.startswith(("http://", "https://")) and \ - 66
not url.startswith(SITE): - 67
host = url.split("/")[2] - 68
if host not in ALLOWED_EXTERNAL: - 69
problems.append(f"{page}: unexpected external host {host}") - 70
continue - 71
fs = local_target(url) - 72
if fs and not os.path.isfile(fs): - 73
problems.append(f"{page}: dead local link {url} -> {fs}") - 74
- 75
# Accessibility / metadata basics - 76
if "<h1" not in src: - 77
problems.append(f"{page}: no <h1>") - 78
if 'name="description"' not in src: - 79
problems.append(f"{page}: no meta description") - 80
if src.count("<title>") != 1: - 81
problems.append(f"{page}: expected exactly one <title>") - 82
if "lang=" not in src.split(">")[1]: - 83
problems.append(f"{page}: <html> missing lang") - 84
- 85
# No page should ship a stray template artefact. Patterns are anchored so - 86
# ordinary prose ("None of this is mysterious…") doesn't trip them. - 87
ARTEFACTS = [ - 88
(r">None<", "bare None rendered into markup"), - 89
(r"=\"None\"", "None in an attribute"), - 90
(r"\bNaN\b", "NaN"), - 91
(r"\bundefined\b", "undefined"), - 92
(r"\{d\[", "unexpanded f-string"), - 93
(r"\{DATA", "unexpanded f-string"), - 94
(r"\{[a-z_]+\[[\"']", "unexpanded f-string"), - 95
(r"\{[A-Z][A-Z_]{2,}\}", "unexpanded template placeholder"), - 96
] - 97
for page in PAGES: - 98
src = open(os.path.join(ROOT, page), encoding="utf-8").read() - 99
for pattern, label in ARTEFACTS: - 100
if re.search(pattern, src): - 101
problems.append(f"{page}: {label} in output") - 102
- 103
# Feed and sitemap: well-formed, and pointing only at pages that exist. - 104
import xml.etree.ElementTree as ET - 105
atom = "{http://www.w3.org/2005/Atom}" - 106
sm = "{http://www.sitemaps.org/schemas/sitemap/0.9}" - 107
try: - 108
feed = ET.parse(os.path.join(ROOT, "feed.xml")).getroot() - 109
entries = feed.findall(atom + "entry") - 110
if not entries: - 111
problems.append("feed.xml: no entries") - 112
for e in entries: - 113
link = e.find(atom + "link") - 114
href = link.get("href") if link is not None else "" - 115
path = href.replace("https://sweedworks.com", "") - 116
target = local_target(path) - 117
if not target or not os.path.isfile(target): - 118
problems.append(f"feed.xml: entry points at missing page {href}") - 119
print(f"PASS feed.xml {len(entries)} entries, all resolve") - 120
except Exception as e: - 121
problems.append(f"feed.xml: {e}") - 122
- 123
try: - 124
smap = ET.parse(os.path.join(ROOT, "sitemap.xml")).getroot() - 125
locs = [u.text for u in smap.iter(sm + "loc")] - 126
for loc in locs: - 127
target = local_target(loc.replace("https://sweedworks.com", "")) - 128
if not target or not os.path.isfile(target): - 129
problems.append(f"sitemap.xml: missing page {loc}") - 130
print(f"PASS sitemap.xml {len(locs)} urls, all resolve") - 131
except Exception as e: - 132
problems.append(f"sitemap.xml: {e}") - 133
- 134
# Every published page should advertise the feed. - 135
for page in PAGES: - 136
src = open(os.path.join(ROOT, page), encoding="utf-8").read() - 137
if 'type="application/atom+xml"' not in src: - 138
problems.append(f"{page}: no feed discovery link") - 139
# /.build/ is denied by the web server, so a link into it is a dead link - 140
# that this checker would otherwise pass, since the file exists on disk. - 141
if 'href="/.build/' in src or 'src="/.build/' in src: - 142
problems.append(f"{page}: links into /.build/, which returns 403") - 143
- 144
print() - 145
if problems: - 146
print(f"{len(problems)} problem(s):") - 147
for p in problems: - 148
print(f" - {p}") - 149
return 1 - 150
print("All pages valid, all local links resolve, no third-party assets.") - 151
return 0 - 152
- 153
- 154
if __name__ == "__main__": - 155
sys.exit(main())