sweedworks

← all sources

checkdelivery.py

What a reader downloads, and whether it caches

117 lines. This is the file the build actually runs, copied verbatim at build time.

  1. 1"""What a reader actually downloads, and whether it stays downloaded.
  2. 2 
  3. 3Two things this site assumes and has never checked:
  4. 4 
  5. 5 * The tokenizer bundles are ~2 MB raw. If they are not compressed in transit,
  6. 6 every reader pays four times what they need to. Nothing in the build knows
  7. 7 whether compression actually happens — that is the CDN's decision, not mine.
  8. 8 * Asset URLs carry a content hash (/style.css?v=abc123) specifically so they
  9. 9 can be cached forever. That is pointless if the cache headers do not say so,
  10. 10 and worse than pointless if HTML is cached too, because then corrections
  11. 11 never reach anyone.
  12. 12 
  13. 13 python3 checkdelivery.py
  14. 14"""
  15. 15 
  16. 16import os
  17. 17import re
  18. 18import subprocess
  19. 19import sys
  20. 20 
  21. 21HERE = os.path.dirname(os.path.abspath(__file__))
  22. 22ROOT = os.path.abspath(os.path.join(HERE, ".."))
  23. 23BASE = "https://sweedworks.com"
  24. 24 
  25. 25ASSETS = [
  26. 26 "/vendor/gpt-tokenizer/o200k_base.js",
  27. 27 "/vendor/gpt-tokenizer/cl100k_base.js",
  28. 28 "/attention/model.js",
  29. 29 "/style.css",
  30. 30 "/learn/mlp.js",
  31. 31 "/feed.xml",
  32. 32]
  33. 33PAGES = ["/", "/tokens/", "/about/"]
  34. 34 
  35. 35BIG = 50_000 # worth compressing
  36. 36LONG_CACHE = 86_400 # a day, in seconds
  37. 37 
  38. 38 
  39. 39def head(url, compressed):
  40. 40 """Headers plus the number of bytes curl actually pulled down.
  41. 41 
  42. 42 content-length is absent on compressed responses, so it cannot be used to
  43. 43 measure transfer. %{size_download} is what really crossed the wire.
  44. 44 """
  45. 45 cmd = ["curl", "-sS", "-o", "/dev/null", "-D", "-",
  46. 46 "-w", "\n__size__:%{size_download}", BASE + url]
  47. 47 if compressed:
  48. 48 cmd.insert(1, "--compressed")
  49. 49 out = subprocess.run(cmd, capture_output=True, text=True, timeout=120)
  50. 50 headers = {}
  51. 51 for line in out.stdout.splitlines():
  52. 52 if line.startswith("__size__:"):
  53. 53 headers["__size__"] = line.split(":", 1)[1].strip()
  54. 54 elif ":" in line:
  55. 55 k, v = line.split(":", 1)
  56. 56 headers[k.strip().lower()] = v.strip()
  57. 57 return headers
  58. 58 
  59. 59 
  60. 60def max_age(cache_control):
  61. 61 m = re.search(r"max-age=(\d+)", cache_control or "")
  62. 62 return int(m.group(1)) if m else None
  63. 63 
  64. 64 
  65. 65def main():
  66. 66 problems, rows = [], []
  67. 67 
  68. 68 for url in ASSETS:
  69. 69 path = os.path.join(ROOT, url.lstrip("/"))
  70. 70 raw = os.path.getsize(path) if os.path.exists(path) else 0
  71. 71 h = head(url, compressed=True)
  72. 72 enc = h.get("content-encoding", "none")
  73. 73 sent = int(h.get("__size__", 0) or 0)
  74. 74 cc = h.get("cache-control", "")
  75. 75 age = max_age(cc)
  76. 76 
  77. 77 rows.append((url, raw, sent, enc, cc or "—"))
  78. 78 
  79. 79 if raw >= BIG and sent and sent > 0.9 * raw:
  80. 80 problems.append(
  81. 81 f"{url}: {raw:,} bytes on disk, {sent:,} received — barely "
  82. 82 f"compressed, readers are paying for the whole file")
  83. 83 if "?v=" not in url and age is not None and age > LONG_CACHE:
  84. 84 # These are unhashed URLs; a long cache means corrections stick.
  85. 85 problems.append(f"{url}: cached for {age:,}s but has no version in "
  86. 86 f"the URL, so a fix cannot reach anyone who has it")
  87. 87 
  88. 88 print(f"{'asset':<40} {'on disk':>10} {'sent':>10} encoding cache-control")
  89. 89 for url, raw, sent, enc, cc in rows:
  90. 90 ratio = f"{100 * sent / raw:.0f}%" if raw and sent else ""
  91. 91 print(f" {url:<38} {raw:>10,} {sent:>10,} {enc:<10} {cc[:36]} {ratio}")
  92. 92 
  93. 93 print()
  94. 94 for url in PAGES:
  95. 95 h = head(url, compressed=True)
  96. 96 cc = h.get("cache-control", "")
  97. 97 age = max_age(cc)
  98. 98 enc = h.get("content-encoding", "none")
  99. 99 print(f" page {url:<12} encoding {enc:<8} cache-control: {cc or '—'}")
  100. 100 if age and age > 3600:
  101. 101 problems.append(f"{url}: HTML cached for {age:,}s — a correction "
  102. 102 f"would not reach readers for that long")
  103. 103 
  104. 104 print()
  105. 105 if problems:
  106. 106 print(f"{len(problems)} problem(s):")
  107. 107 for p in problems:
  108. 108 print(f" - {p}")
  109. 109 return 1
  110. 110 print("Assets are compressed and cacheable; HTML is not cached long enough "
  111. 111 "to trap a correction.")
  112. 112 return 0
  113. 113 
  114. 114 
  115. 115if __name__ == "__main__":
  116. 116 sys.exit(main())