checkdelivery.py
What a reader downloads, and whether it caches
117 lines. This is the file the build actually runs, copied verbatim at build time.
- 1
"""What a reader actually downloads, and whether it stays downloaded. - 2
- 3
Two things this site assumes and has never checked: - 4
- 5
* The tokenizer bundles are ~2 MB raw. If they are not compressed in transit, - 6
every reader pays four times what they need to. Nothing in the build knows - 7
whether compression actually happens — that is the CDN's decision, not mine. - 8
* Asset URLs carry a content hash (/style.css?v=abc123) specifically so they - 9
can be cached forever. That is pointless if the cache headers do not say so, - 10
and worse than pointless if HTML is cached too, because then corrections - 11
never reach anyone. - 12
- 13
python3 checkdelivery.py - 14
""" - 15
- 16
import os - 17
import re - 18
import subprocess - 19
import sys - 20
- 21
HERE = os.path.dirname(os.path.abspath(__file__)) - 22
ROOT = os.path.abspath(os.path.join(HERE, "..")) - 23
BASE = "https://sweedworks.com" - 24
- 25
ASSETS = [ - 26
"/vendor/gpt-tokenizer/o200k_base.js", - 27
"/vendor/gpt-tokenizer/cl100k_base.js", - 28
"/attention/model.js", - 29
"/style.css", - 30
"/learn/mlp.js", - 31
"/feed.xml", - 32
] - 33
PAGES = ["/", "/tokens/", "/about/"] - 34
- 35
BIG = 50_000 # worth compressing - 36
LONG_CACHE = 86_400 # a day, in seconds - 37
- 38
- 39
def head(url, compressed): - 40
"""Headers plus the number of bytes curl actually pulled down. - 41
- 42
content-length is absent on compressed responses, so it cannot be used to - 43
measure transfer. %{size_download} is what really crossed the wire. - 44
""" - 45
cmd = ["curl", "-sS", "-o", "/dev/null", "-D", "-", - 46
"-w", "\n__size__:%{size_download}", BASE + url] - 47
if compressed: - 48
cmd.insert(1, "--compressed") - 49
out = subprocess.run(cmd, capture_output=True, text=True, timeout=120) - 50
headers = {} - 51
for line in out.stdout.splitlines(): - 52
if line.startswith("__size__:"): - 53
headers["__size__"] = line.split(":", 1)[1].strip() - 54
elif ":" in line: - 55
k, v = line.split(":", 1) - 56
headers[k.strip().lower()] = v.strip() - 57
return headers - 58
- 59
- 60
def max_age(cache_control): - 61
m = re.search(r"max-age=(\d+)", cache_control or "") - 62
return int(m.group(1)) if m else None - 63
- 64
- 65
def main(): - 66
problems, rows = [], [] - 67
- 68
for url in ASSETS: - 69
path = os.path.join(ROOT, url.lstrip("/")) - 70
raw = os.path.getsize(path) if os.path.exists(path) else 0 - 71
h = head(url, compressed=True) - 72
enc = h.get("content-encoding", "none") - 73
sent = int(h.get("__size__", 0) or 0) - 74
cc = h.get("cache-control", "") - 75
age = max_age(cc) - 76
- 77
rows.append((url, raw, sent, enc, cc or "—")) - 78
- 79
if raw >= BIG and sent and sent > 0.9 * raw: - 80
problems.append( - 81
f"{url}: {raw:,} bytes on disk, {sent:,} received — barely " - 82
f"compressed, readers are paying for the whole file") - 83
if "?v=" not in url and age is not None and age > LONG_CACHE: - 84
# These are unhashed URLs; a long cache means corrections stick. - 85
problems.append(f"{url}: cached for {age:,}s but has no version in " - 86
f"the URL, so a fix cannot reach anyone who has it") - 87
- 88
print(f"{'asset':<40} {'on disk':>10} {'sent':>10} encoding cache-control") - 89
for url, raw, sent, enc, cc in rows: - 90
ratio = f"{100 * sent / raw:.0f}%" if raw and sent else "" - 91
print(f" {url:<38} {raw:>10,} {sent:>10,} {enc:<10} {cc[:36]} {ratio}") - 92
- 93
print() - 94
for url in PAGES: - 95
h = head(url, compressed=True) - 96
cc = h.get("cache-control", "") - 97
age = max_age(cc) - 98
enc = h.get("content-encoding", "none") - 99
print(f" page {url:<12} encoding {enc:<8} cache-control: {cc or '—'}") - 100
if age and age > 3600: - 101
problems.append(f"{url}: HTML cached for {age:,}s — a correction " - 102
f"would not reach readers for that long") - 103
- 104
print() - 105
if problems: - 106
print(f"{len(problems)} problem(s):") - 107
for p in problems: - 108
print(f" - {p}") - 109
return 1 - 110
print("Assets are compressed and cacheable; HTML is not cached long enough " - 111
"to trap a correction.") - 112
return 0 - 113
- 114
- 115
if __name__ == "__main__": - 116
sys.exit(main())