sweedworks

← all sources

checkrequests.py

Every host a browser really contacted

117 lines. This is the file the build actually runs, copied verbatim at build time.

  1. 1"""Every host a browser actually requested, parsed from Chrome's network log.
  2. 2 
  3. 3The site claims "no third-party requests" on every page. That was only checked
  4. 4by grepping the generated HTML for foreign hostnames, which cannot see what a
  5. 5script fetches at run time — and Cloudflare injects a script into every page.
  6. 6 
  7. 7A warning learned the hard way: do NOT grep URLs out of the netlog. The file
  8. 8contains Chrome's own configuration — the HSTS preload list, safe-browsing and
  9. 9autofill tables — so a regex over it "finds" youtube.com and play.google.com on
  10. 10a page that never touched them. That would have been a confident, wrong,
  11. 11alarming claim. This parses real request events instead.
  12. 12 
  13. 13Capture and analysis are separate: only shot.sh may reach the network and start
  14. 14chrome, so run it per page first, then
  15. 15 
  16. 16 python3 checkrequests.py /tmp/claude-996/net-*.json
  17. 17"""
  18. 18 
  19. 19import json
  20. 20import os
  21. 21import re
  22. 22import sys
  23. 23 
  24. 24OWN = {"sweedworks.com", "www.sweedworks.com"}
  25. 25 
  26. 26# Hosts Chrome contacts on its own account at startup. Verified by capturing a
  27. 27# load of /robots.txt — a plain text file with no HTML, CSS or scripts — and
  28. 28# seeing exactly these. They would appear on any site in the world. Listed
  29. 29# rather than filtered silently, and deliberately narrow: a real request to a
  30. 30# Google host from page code would still be reported.
  31. 31BROWSER_STARTUP = {
  32. 32 "accounts.google.com",
  33. 33 "www.google.com",
  34. 34 "clients2.google.com",
  35. 35 "content-autofill.googleapis.com",
  36. 36}
  37. 37HOST_RE = re.compile(r"^https?://([^/:]+)")
  38. 38 
  39. 39# Event types that mean a request was actually started.
  40. 40REQUEST_EVENTS = {"URL_REQUEST_START_JOB", "HTTP_STREAM_JOB_CONTROLLER_BOUND"}
  41. 41 
  42. 42 
  43. 43def analyse(path):
  44. 44 with open(path, encoding="utf-8", errors="replace") as fh:
  45. 45 raw = fh.read()
  46. 46 
  47. 47 # Chrome may not close the JSON if it was killed; salvage what parses.
  48. 48 try:
  49. 49 log = json.loads(raw)
  50. 50 except ValueError:
  51. 51 cut = raw.rfind("},")
  52. 52 log = json.loads(raw[:cut + 1] + "]}")
  53. 53 
  54. 54 types = log.get("constants", {}).get("logEventTypes", {})
  55. 55 wanted = {tid for name, tid in types.items() if name in REQUEST_EVENTS}
  56. 56 if not wanted:
  57. 57 raise RuntimeError(f"{path}: no request event types in constants")
  58. 58 
  59. 59 requested = []
  60. 60 for ev in log.get("events", []):
  61. 61 if ev.get("type") not in wanted:
  62. 62 continue
  63. 63 url = (ev.get("params") or {}).get("url")
  64. 64 if url:
  65. 65 requested.append(url)
  66. 66 
  67. 67 own, foreign, browser = set(), {}, {}
  68. 68 for url in requested:
  69. 69 m = HOST_RE.match(url)
  70. 70 if not m:
  71. 71 continue
  72. 72 host = m.group(1).lower()
  73. 73 if host in OWN:
  74. 74 own.add(url.split("?")[0])
  75. 75 elif host in BROWSER_STARTUP:
  76. 76 browser[host] = browser.get(host, 0) + 1
  77. 77 else:
  78. 78 foreign[host] = foreign.get(host, 0) + 1
  79. 79 return own, foreign, browser, len(requested)
  80. 80 
  81. 81 
  82. 82def main():
  83. 83 logs = sys.argv[1:]
  84. 84 if not logs:
  85. 85 print(__doc__)
  86. 86 return 2
  87. 87 
  88. 88 problems = []
  89. 89 for path in logs:
  90. 90 if not os.path.exists(path):
  91. 91 problems.append(f"{path}: missing — was shot.sh run for it?")
  92. 92 continue
  93. 93 own, foreign, browser, total = analyse(path)
  94. 94 print(f"{os.path.basename(path)} ({total} requests started)")
  95. 95 for u in sorted(own):
  96. 96 tag = "cloudflare" if "/cdn-cgi/" in u else "own"
  97. 97 print(f" {tag:<10} {u}")
  98. 98 for h, n in sorted(browser.items()):
  99. 99 print(f" browser {h} ({n})")
  100. 100 for h, n in sorted(foreign.items()):
  101. 101 print(f" FOREIGN {h} ({n})")
  102. 102 problems.append(f"{os.path.basename(path)}: requested {h}")
  103. 103 print()
  104. 104 
  105. 105 if problems:
  106. 106 print(f"{len(problems)} problem(s):")
  107. 107 for p in problems:
  108. 108 print(f" - {p}")
  109. 109 return 1
  110. 110 print("Every request from page code went to sweedworks.com. The only other "
  111. 111 "traffic is Chrome's own startup calls, which happen on any site.")
  112. 112 return 0
  113. 113 
  114. 114 
  115. 115if __name__ == "__main__":
  116. 116 sys.exit(main())