sweedworks

← all sources

checkhtml.py

Strict HTML parse, dead links, feed and sitemap

156 lines. This is the file the build actually runs, copied verbatim at build time.

  1. 1"""Parse every generated page with html5lib in strict mode and report problems.
  2. 2 
  3. 3Also does a few checks a parser won't: internal links resolve to real files,
  4. 4referenced assets exist, and no page accidentally references a third-party host.
  5. 5"""
  6. 6 
  7. 7import os
  8. 8import re
  9. 9import sys
  10. 10 
  11. 11HERE = os.path.dirname(os.path.abspath(__file__))
  12. 12sys.path.insert(0, os.path.join(HERE, "pylib"))
  13. 13ROOT = os.path.abspath(os.path.join(HERE, ".."))
  14. 14 
  15. 15import html5lib # noqa: E402
  16. 16 
  17. 17PAGES = ["index.html", "tokens/index.html", "vocabulary/index.html",
  18. 18 "predict/index.html", "cost/index.html", "learn/index.html", "attention/index.html",
  19. 19 "about/index.html", "changes/index.html", "checking/index.html"]
  20. 20 
  21. 21# Hosts a page is allowed to link to in href/src. Anything else is a surprise.
  22. 22ALLOWED_EXTERNAL = {"github.com"}
  23. 23 
  24. 24SITE = "https://sweedworks.com"
  25. 25 
  26. 26 
  27. 27def local_target(url):
  28. 28 """Map a site-absolute URL to a file on disk, or None if not local.
  29. 29 
  30. 30 Absolute URLs to our own domain count as local: og:image, og:url and
  31. 31 canonical all have to be absolute, and they still need to resolve.
  32. 32 """
  33. 33 if url.startswith(SITE):
  34. 34 url = url[len(SITE):] or "/"
  35. 35 if url.startswith(("http://", "https://", "mailto:", "#", "data:")):
  36. 36 return None
  37. 37 path = url.split("#")[0].split("?")[0]
  38. 38 if not path.startswith("/"):
  39. 39 return None
  40. 40 fs = os.path.join(ROOT, path.lstrip("/"))
  41. 41 if path.endswith("/"):
  42. 42 fs = os.path.join(fs, "index.html")
  43. 43 return fs
  44. 44 
  45. 45 
  46. 46def main():
  47. 47 problems = []
  48. 48 
  49. 49 for page in PAGES:
  50. 50 full = os.path.join(ROOT, page)
  51. 51 src = open(full, encoding="utf-8").read()
  52. 52 
  53. 53 parser = html5lib.HTMLParser(strict=True)
  54. 54 try:
  55. 55 parser.parse(src)
  56. 56 print(f"PASS {page:<20} parses clean")
  57. 57 except Exception as e:
  58. 58 problems.append(f"{page}: parse error: {e}")
  59. 59 print(f"FAIL {page:<20} {str(e)[:120]}")
  60. 60 continue
  61. 61 
  62. 62 # Links and assets
  63. 63 for attr in ("href", "src", "content"):
  64. 64 for url in re.findall(rf'{attr}="([^"]+)"', src):
  65. 65 if url.startswith(("http://", "https://")) and \
  66. 66 not url.startswith(SITE):
  67. 67 host = url.split("/")[2]
  68. 68 if host not in ALLOWED_EXTERNAL:
  69. 69 problems.append(f"{page}: unexpected external host {host}")
  70. 70 continue
  71. 71 fs = local_target(url)
  72. 72 if fs and not os.path.isfile(fs):
  73. 73 problems.append(f"{page}: dead local link {url} -> {fs}")
  74. 74 
  75. 75 # Accessibility / metadata basics
  76. 76 if "<h1" not in src:
  77. 77 problems.append(f"{page}: no <h1>")
  78. 78 if 'name="description"' not in src:
  79. 79 problems.append(f"{page}: no meta description")
  80. 80 if src.count("<title>") != 1:
  81. 81 problems.append(f"{page}: expected exactly one <title>")
  82. 82 if "lang=" not in src.split(">")[1]:
  83. 83 problems.append(f"{page}: <html> missing lang")
  84. 84 
  85. 85 # No page should ship a stray template artefact. Patterns are anchored so
  86. 86 # ordinary prose ("None of this is mysterious…") doesn't trip them.
  87. 87 ARTEFACTS = [
  88. 88 (r">None<", "bare None rendered into markup"),
  89. 89 (r"=\"None\"", "None in an attribute"),
  90. 90 (r"\bNaN\b", "NaN"),
  91. 91 (r"\bundefined\b", "undefined"),
  92. 92 (r"\{d\[", "unexpanded f-string"),
  93. 93 (r"\{DATA", "unexpanded f-string"),
  94. 94 (r"\{[a-z_]+\[[\"']", "unexpanded f-string"),
  95. 95 (r"\{[A-Z][A-Z_]{2,}\}", "unexpanded template placeholder"),
  96. 96 ]
  97. 97 for page in PAGES:
  98. 98 src = open(os.path.join(ROOT, page), encoding="utf-8").read()
  99. 99 for pattern, label in ARTEFACTS:
  100. 100 if re.search(pattern, src):
  101. 101 problems.append(f"{page}: {label} in output")
  102. 102 
  103. 103 # Feed and sitemap: well-formed, and pointing only at pages that exist.
  104. 104 import xml.etree.ElementTree as ET
  105. 105 atom = "{http://www.w3.org/2005/Atom}"
  106. 106 sm = "{http://www.sitemaps.org/schemas/sitemap/0.9}"
  107. 107 try:
  108. 108 feed = ET.parse(os.path.join(ROOT, "feed.xml")).getroot()
  109. 109 entries = feed.findall(atom + "entry")
  110. 110 if not entries:
  111. 111 problems.append("feed.xml: no entries")
  112. 112 for e in entries:
  113. 113 link = e.find(atom + "link")
  114. 114 href = link.get("href") if link is not None else ""
  115. 115 path = href.replace("https://sweedworks.com", "")
  116. 116 target = local_target(path)
  117. 117 if not target or not os.path.isfile(target):
  118. 118 problems.append(f"feed.xml: entry points at missing page {href}")
  119. 119 print(f"PASS feed.xml {len(entries)} entries, all resolve")
  120. 120 except Exception as e:
  121. 121 problems.append(f"feed.xml: {e}")
  122. 122 
  123. 123 try:
  124. 124 smap = ET.parse(os.path.join(ROOT, "sitemap.xml")).getroot()
  125. 125 locs = [u.text for u in smap.iter(sm + "loc")]
  126. 126 for loc in locs:
  127. 127 target = local_target(loc.replace("https://sweedworks.com", ""))
  128. 128 if not target or not os.path.isfile(target):
  129. 129 problems.append(f"sitemap.xml: missing page {loc}")
  130. 130 print(f"PASS sitemap.xml {len(locs)} urls, all resolve")
  131. 131 except Exception as e:
  132. 132 problems.append(f"sitemap.xml: {e}")
  133. 133 
  134. 134 # Every published page should advertise the feed.
  135. 135 for page in PAGES:
  136. 136 src = open(os.path.join(ROOT, page), encoding="utf-8").read()
  137. 137 if 'type="application/atom+xml"' not in src:
  138. 138 problems.append(f"{page}: no feed discovery link")
  139. 139 # /.build/ is denied by the web server, so a link into it is a dead link
  140. 140 # that this checker would otherwise pass, since the file exists on disk.
  141. 141 if 'href="/.build/' in src or 'src="/.build/' in src:
  142. 142 problems.append(f"{page}: links into /.build/, which returns 403")
  143. 143 
  144. 144 print()
  145. 145 if problems:
  146. 146 print(f"{len(problems)} problem(s):")
  147. 147 for p in problems:
  148. 148 print(f" - {p}")
  149. 149 return 1
  150. 150 print("All pages valid, all local links resolve, no third-party assets.")
  151. 151 return 0
  152. 152 
  153. 153 
  154. 154if __name__ == "__main__":
  155. 155 sys.exit(main())