sweedworks

← all sources

checkfigures.py

No number in the prose that is not in the data

110 lines. This is the file the build actually runs, copied verbatim at build time.

  1. 1"""Find numbers typed into the prose by hand.
  2. 2 
  3. 3The site's central claim is that every figure is computed rather than written
  4. 4down. That is easy to believe and easy to break: one edit, one remembered
  5. 5number, and a page states something that no longer follows from the data.
  6. 6 
  7. 7This scans the rendered HTML for numeric-looking claims and checks each against
  8. 8the precomputed data files. Numbers that appear in the data are fine. Numbers
  9. 9that do not are listed for a human decision — many are legitimate (a year, a
  10. 10pixel size, "four merges"), so this reports rather than fails.
  11. 11 
  12. 12 python3 checkfigures.py
  13. 13"""
  14. 14 
  15. 15import json
  16. 16import os
  17. 17import re
  18. 18import sys
  19. 19 
  20. 20HERE = os.path.dirname(os.path.abspath(__file__))
  21. 21ROOT = os.path.abspath(os.path.join(HERE, ".."))
  22. 22 
  23. 23PAGES = ["tokens/index.html", "vocabulary/index.html", "predict/index.html",
  24. 24 "cost/index.html", "learn/index.html", "attention/index.html"]
  25. 25 
  26. 26DATA_FILES = ["tokens/data.json", "vocabulary/data.json", "predict/data.json",
  27. 27 "cost/data.json", "learn/data.json", "attention/data.json"]
  28. 28 
  29. 29# Figures worth auditing: thousands separators, percentages, multipliers,
  30. 30# decimals. Bare small integers are almost always prose ("four merges").
  31. 31FIGURE_RE = re.compile(
  32. 32 r"\b(\d{1,3}(?:,\d{3})+|\d+\.\d+(?:×|%)?|\d{2,}%|\d+×)")
  33. 33 
  34. 34TAG_RE = re.compile(r"<[^>]+>")
  35. 35 
  36. 36 
  37. 37def data_numbers():
  38. 38 """Every number that appears anywhere in the precomputed data."""
  39. 39 seen = set()
  40. 40 
  41. 41 def walk(node):
  42. 42 if isinstance(node, dict):
  43. 43 for v in node.values():
  44. 44 walk(v)
  45. 45 elif isinstance(node, list):
  46. 46 for v in node:
  47. 47 walk(v)
  48. 48 elif isinstance(node, str):
  49. 49 # Inputs and sample text are data too: a number being tokenized is
  50. 50 # not a claim about the world.
  51. 51 seen.add(node)
  52. 52 for token in re.findall(r"[\d.,]+", node):
  53. 53 seen.add(token)
  54. 54 elif isinstance(node, bool):
  55. 55 return
  56. 56 elif isinstance(node, (int, float)):
  57. 57 seen.add(f"{node}")
  58. 58 seen.add(f"{node:,}")
  59. 59 if isinstance(node, float):
  60. 60 for places in (0, 1, 2):
  61. 61 seen.add(f"{node:.{places}f}")
  62. 62 seen.add(f"{100 * node:.{places}f}")
  63. 63 seen.add(f"{round(100 * node)}")
  64. 64 else:
  65. 65 seen.add(f"{node // 1000},000")
  66. 66 
  67. 67 for rel in DATA_FILES:
  68. 68 path = os.path.join(ROOT, rel)
  69. 69 if os.path.exists(path):
  70. 70 walk(json.load(open(path, encoding="utf-8")))
  71. 71 return seen
  72. 72 
  73. 73 
  74. 74def main():
  75. 75 known = data_numbers()
  76. 76 # Values that are structural rather than findings.
  77. 77 allowed = {"1,000", "200,000", "100,000", "1200", "630", "2026", "1859",
  78. 78 "4.5", "3.0", "0.0", "1.0", "2.0", "0.5", "1.5", "2.5"}
  79. 79 
  80. 80 unexplained = {}
  81. 81 for page in PAGES:
  82. 82 src = open(os.path.join(ROOT, page), encoding="utf-8").read()
  83. 83 text = TAG_RE.sub(" ", src)
  84. 84 for m in FIGURE_RE.finditer(text):
  85. 85 fig = m.group(1)
  86. 86 bare = fig.rstrip("×%")
  87. 87 if fig in known or bare in known or fig in allowed or bare in allowed:
  88. 88 continue
  89. 89 start = max(0, m.start() - 60)
  90. 90 ctx = " ".join(text[start:m.end() + 40].split())
  91. 91 unexplained.setdefault(page, []).append((fig, ctx))
  92. 92 
  93. 93 total = sum(len(v) for v in unexplained.values())
  94. 94 if not total:
  95. 95 print("Every figure in the prose matches a value in the computed data.")
  96. 96 return 0
  97. 97 
  98. 98 print(f"{total} figure(s) not found in the computed data — check each:")
  99. 99 for page, items in unexplained.items():
  100. 100 print(f"\n {page}")
  101. 101 for fig, ctx in items:
  102. 102 print(f" {fig:<10} …{ctx}…")
  103. 103 print("\nSome will be legitimate prose. Any that are real claims should be "
  104. 104 "interpolated from the data instead.")
  105. 105 return 0
  106. 106 
  107. 107 
  108. 108if __name__ == "__main__":
  109. 109 sys.exit(main())