checklive.py
Compares served bytes against what was generated
139 lines. This is the file the build actually runs, copied verbatim at build time.
- 1
"""Compare what I generate against what is actually served. - 2
- 3
Cloudflare injects a bot-detection script into HTML in transit. I missed it for - 4
days because I only ever checked the files I wrote — the injection is invisible - 5
from disk. This script diffs the served bytes against the local file and reports - 6
anything added, so a change in what sits in front of this domain shows up as a - 7
build finding rather than as a false claim on the privacy page. - 8
- 9
Needs real network access to the live domain, so it is not part of build.sh: - 10
run it after deploying. - 11
python3 checklive.py - 12
""" - 13
- 14
import os - 15
import re - 16
import subprocess - 17
import sys - 18
- 19
HERE = os.path.dirname(os.path.abspath(__file__)) - 20
ROOT = os.path.abspath(os.path.join(HERE, "..")) - 21
BASE = "https://sweedworks.com" - 22
- 23
PAGES = [("/", "index.html"), - 24
("/vocabulary/", "vocabulary/index.html"), - 25
("/predict/", "predict/index.html"), - 26
("/cost/", "cost/index.html"), - 27
("/learn/", "learn/index.html"), - 28
("/attention/", "attention/index.html"), - 29
("/tokens/", "tokens/index.html"), - 30
("/about/", "about/index.html"), - 31
("/changes/", "changes/index.html"), - 32
("/checking/", "checking/index.html")] - 33
- 34
# Injections we already know about and have disclosed on /about/. - 35
KNOWN = [ - 36
(r"__CF\$cv\$params", "Cloudflare bot-detection script (disclosed)"), - 37
(r"cdn-cgi/challenge-platform", "Cloudflare challenge platform (disclosed)"), - 38
] - 39
- 40
- 41
def fetch(path): - 42
"""Headers and body to separate files. - 43
- 44
Do not go back to `-D -` with text=True: universal-newline translation turns - 45
the CRLF header/body separator into LF, the partition silently fails, the - 46
body comes back empty, and every comparison below then "passes" against - 47
nothing. This check reported all-clear that way while the injected script - 48
was plainly there. - 49
""" - 50
import tempfile - 51
with tempfile.TemporaryDirectory() as tmp: - 52
hp = os.path.join(tmp, "h") - 53
bp = os.path.join(tmp, "b") - 54
subprocess.run( - 55
["curl", "-sS", "-D", hp, "-o", bp, f"{BASE}{path}?livecheck=1"], - 56
capture_output=True, timeout=60, check=True) - 57
head = open(hp, encoding="utf-8", errors="replace").read() - 58
body = open(bp, encoding="utf-8", errors="replace").read() - 59
if not body: - 60
raise RuntimeError(f"empty body for {path} — the check would be vacuous") - 61
return head, body - 62
- 63
- 64
def main(): - 65
problems, notes = [], [] - 66
- 67
for url, local in PAGES: - 68
head, served = fetch(url) - 69
mine = open(os.path.join(ROOT, local), encoding="utf-8").read() - 70
- 71
# Anything in the served copy that is not in mine. - 72
extra = served - 73
for line in mine.splitlines(): - 74
extra = extra.replace(line, "", 1) - 75
extra = extra.strip() - 76
- 77
if extra: - 78
explained = False - 79
for pattern, label in KNOWN: - 80
if re.search(pattern, extra): - 81
notes.append(f"{url}: {label}, {len(extra):,} bytes added") - 82
explained = True - 83
break - 84
if not explained: - 85
problems.append( - 86
f"{url}: UNEXPLAINED content injected in transit " - 87
f"({len(extra):,} bytes): {extra[:200]!r}") - 88
else: - 89
notes.append(f"{url}: served bytes match what I generated") - 90
- 91
if re.search(r"(?im)^set-cookie:", head): - 92
problems.append(f"{url}: a cookie is being set — /about/ says none are") - 93
- 94
for m in re.finditer(r"(?im)^(nel|report-to):\s*(.*)$", head): - 95
if "cloudflare" in m.group(2).lower(): - 96
notes.append(f"{url}: {m.group(1)} header points at Cloudflare " - 97
f"(disclosed)") - 98
- 99
# Every internal link must actually be reachable. Checking that files exist - 100
# on disk is not enough: the web server denies some paths, so a link can be - 101
# perfectly valid locally and 403 to the public. That happened. - 102
seen, checked = set(), 0 - 103
for url, local in PAGES: - 104
src = open(os.path.join(ROOT, local), encoding="utf-8").read() - 105
for attr in ("href", "src"): - 106
for target in re.findall(rf'{attr}="([^"]+)"', src): - 107
if target.startswith(("http://", "https://", "#", "mailto:", - 108
"data:")): - 109
continue - 110
clean = target.split("#")[0] - 111
if not clean.startswith("/") or clean in seen: - 112
continue - 113
seen.add(clean) - 114
out = subprocess.run( - 115
["curl", "-sS", "-o", "/dev/null", "-w", "%{http_code}", - 116
f"{BASE}{clean}"], - 117
capture_output=True, text=True, timeout=60) - 118
code = out.stdout.strip() - 119
checked += 1 - 120
if code != "200": - 121
problems.append(f"{url}: links to {clean} which returns {code}") - 122
notes.append(f"{checked} internal links checked, all reachable" - 123
if not problems else f"{checked} internal links checked") - 124
- 125
for n in notes: - 126
print(f" note {n}") - 127
print() - 128
if problems: - 129
print(f"{len(problems)} problem(s):") - 130
for p in problems: - 131
print(f" - {p}") - 132
return 1 - 133
print("Live pages match what I generated, apart from disclosed injections.") - 134
return 0 - 135
- 136
- 137
if __name__ == "__main__": - 138
sys.exit(main())