checkfigures.py
No number in the prose that is not in the data
110 lines. This is the file the build actually runs, copied verbatim at build time.
- 1
"""Find numbers typed into the prose by hand. - 2
- 3
The site's central claim is that every figure is computed rather than written - 4
down. That is easy to believe and easy to break: one edit, one remembered - 5
number, and a page states something that no longer follows from the data. - 6
- 7
This scans the rendered HTML for numeric-looking claims and checks each against - 8
the precomputed data files. Numbers that appear in the data are fine. Numbers - 9
that do not are listed for a human decision — many are legitimate (a year, a - 10
pixel size, "four merges"), so this reports rather than fails. - 11
- 12
python3 checkfigures.py - 13
""" - 14
- 15
import json - 16
import os - 17
import re - 18
import sys - 19
- 20
HERE = os.path.dirname(os.path.abspath(__file__)) - 21
ROOT = os.path.abspath(os.path.join(HERE, "..")) - 22
- 23
PAGES = ["tokens/index.html", "vocabulary/index.html", "predict/index.html", - 24
"cost/index.html", "learn/index.html", "attention/index.html"] - 25
- 26
DATA_FILES = ["tokens/data.json", "vocabulary/data.json", "predict/data.json", - 27
"cost/data.json", "learn/data.json", "attention/data.json"] - 28
- 29
# Figures worth auditing: thousands separators, percentages, multipliers, - 30
# decimals. Bare small integers are almost always prose ("four merges"). - 31
FIGURE_RE = re.compile( - 32
r"\b(\d{1,3}(?:,\d{3})+|\d+\.\d+(?:×|%)?|\d{2,}%|\d+×)") - 33
- 34
TAG_RE = re.compile(r"<[^>]+>") - 35
- 36
- 37
def data_numbers(): - 38
"""Every number that appears anywhere in the precomputed data.""" - 39
seen = set() - 40
- 41
def walk(node): - 42
if isinstance(node, dict): - 43
for v in node.values(): - 44
walk(v) - 45
elif isinstance(node, list): - 46
for v in node: - 47
walk(v) - 48
elif isinstance(node, str): - 49
# Inputs and sample text are data too: a number being tokenized is - 50
# not a claim about the world. - 51
seen.add(node) - 52
for token in re.findall(r"[\d.,]+", node): - 53
seen.add(token) - 54
elif isinstance(node, bool): - 55
return - 56
elif isinstance(node, (int, float)): - 57
seen.add(f"{node}") - 58
seen.add(f"{node:,}") - 59
if isinstance(node, float): - 60
for places in (0, 1, 2): - 61
seen.add(f"{node:.{places}f}") - 62
seen.add(f"{100 * node:.{places}f}") - 63
seen.add(f"{round(100 * node)}") - 64
else: - 65
seen.add(f"{node // 1000},000") - 66
- 67
for rel in DATA_FILES: - 68
path = os.path.join(ROOT, rel) - 69
if os.path.exists(path): - 70
walk(json.load(open(path, encoding="utf-8"))) - 71
return seen - 72
- 73
- 74
def main(): - 75
known = data_numbers() - 76
# Values that are structural rather than findings. - 77
allowed = {"1,000", "200,000", "100,000", "1200", "630", "2026", "1859", - 78
"4.5", "3.0", "0.0", "1.0", "2.0", "0.5", "1.5", "2.5"} - 79
- 80
unexplained = {} - 81
for page in PAGES: - 82
src = open(os.path.join(ROOT, page), encoding="utf-8").read() - 83
text = TAG_RE.sub(" ", src) - 84
for m in FIGURE_RE.finditer(text): - 85
fig = m.group(1) - 86
bare = fig.rstrip("×%") - 87
if fig in known or bare in known or fig in allowed or bare in allowed: - 88
continue - 89
start = max(0, m.start() - 60) - 90
ctx = " ".join(text[start:m.end() + 40].split()) - 91
unexplained.setdefault(page, []).append((fig, ctx)) - 92
- 93
total = sum(len(v) for v in unexplained.values()) - 94
if not total: - 95
print("Every figure in the prose matches a value in the computed data.") - 96
return 0 - 97
- 98
print(f"{total} figure(s) not found in the computed data — check each:") - 99
for page, items in unexplained.items(): - 100
print(f"\n {page}") - 101
for fig, ctx in items: - 102
print(f" {fig:<10} …{ctx}…") - 103
print("\nSome will be legitimate prose. Any that are real claims should be " - 104
"interpolated from the data instead.") - 105
return 0 - 106
- 107
- 108
if __name__ == "__main__": - 109
sys.exit(main())