precompute.py
Computes the figures on /tokens/
147 lines. This is the file the build actually runs, copied verbatim at build time.
- 1
"""Generate tokens/data.json — every number shown on the page is computed here, - 2
never hand-written, so the prose can't drift from the tokenizer.""" - 3
- 4
import json - 5
import os - 6
- 7
from cjsload import Reference - 8
- 9
OUT = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "tokens", "data.json") - 10
- 11
# Article 1 of the Universal Declaration of Human Rights, in the UN's official - 12
# translations. Chosen because it's the same meaning in every row — which is the - 13
# whole point: identical content, wildly different token cost. - 14
UDHR = [ - 15
("English", "en", "Latin", - 16
"All human beings are born free and equal in dignity and rights. They are " - 17
"endowed with reason and conscience and should act towards one another in a " - 18
"spirit of brotherhood."), - 19
("Spanish", "es", "Latin", - 20
"Todos los seres humanos nacen libres e iguales en dignidad y derechos y, " - 21
"dotados como están de razón y conciencia, deben comportarse fraternalmente " - 22
"los unos con los otros."), - 23
("French", "fr", "Latin", - 24
"Tous les êtres humains naissent libres et égaux en dignité et en droits. " - 25
"Ils sont doués de raison et de conscience et doivent agir les uns envers " - 26
"les autres dans un esprit de fraternité."), - 27
("German", "de", "Latin", - 28
"Alle Menschen sind frei und gleich an Würde und Rechten geboren. Sie sind " - 29
"mit Vernunft und Gewissen begabt und sollen einander im Geiste der " - 30
"Brüderlichkeit begegnen."), - 31
("Portuguese", "pt", "Latin", - 32
"Todos os seres humanos nascem livres e iguais em dignidade e direitos. São " - 33
"dotados de razão e consciência e devem agir uns para com os outros em " - 34
"espírito de fraternidade."), - 35
("Russian", "ru", "Cyrillic", - 36
"Все люди рождаются свободными и равными в своем достоинстве и правах. Они " - 37
"наделены разумом и совестью и должны поступать в отношении друг друга в " - 38
"духе братства."), - 39
("Chinese", "zh", "Han", - 40
"人人生而自由,在尊严和权利上一律平等。他们赋有理性和良心,并应以兄弟关系的精神相对待。"), - 41
("Japanese", "ja", "Japanese", - 42
"すべての人間は、生れながらにして自由であり、かつ、尊厳と権利とについて平等である。" - 43
"人間は、理性と良心とを授けられており、互いに同胞の精神をもって行動しなければならない。"), - 44
("Korean", "ko", "Hangul", - 45
"모든 인간은 태어날 때부터 자유로우며 그 존엄과 권리에 있어 동등하다. 인간은 천부적으로 " - 46
"이성과 양심을 부여받았으며 서로 형제애의 정신으로 행동하여야 한다."), - 47
("Arabic", "ar", "Arabic", - 48
"يولد جميع الناس أحراراً متساوين في الكرامة والحقوق. وقد وهبوا عقلاً وضميراً " - 49
"وعليهم أن يعامل بعضهم بعضاً بروح الإخاء."), - 50
("Hindi", "hi", "Devanagari", - 51
"सभी मनुष्यों को गौरव और अधिकारों के मामले में जन्मजात स्वतन्त्रता और समानता प्राप्त है। " - 52
"उन्हें बुद्धि और अन्तरात्मा की देन प्राप्त है और परस्पर उन्हें भाईचारे के भाव से बर्ताव करना चाहिए।"), - 53
] - 54
- 55
COUNTING_WORDS = ["strawberry", "raspberry", "bookkeeper", "Mississippi", "unsuccessfully"] - 56
- 57
NUMBERS = ["1234567890", "1,234,567,890", "3.14159265", "2024", "20240811", "127.0.0.1"] - 58
- 59
WHITESPACE = ["strawberry", " strawberry", "strawberry ", "Strawberry", "STRAWBERRY", " strawberry"] - 60
- 61
CODE = 'def total(items):\n return sum(i.price for i in items)\n' - 62
- 63
PROSE = ( - 64
"The tokenizer does not know what a word is. It knows which byte sequences " - 65
"showed up together often enough during training to deserve their own number." - 66
) - 67
- 68
- 69
def letter_story(tok, word, letter): - 70
"""How the model sees a word it's being asked to spell.""" - 71
toks = tok.tokens(word) - 72
return { - 73
"word": word, - 74
"letter": letter, - 75
"true_count": word.lower().count(letter), - 76
"tokens": toks, - 77
"token_count": len(toks), - 78
} - 79
- 80
- 81
def main(): - 82
o200k = Reference("o200k_base") - 83
cl100k = Reference("cl100k_base") - 84
- 85
data = { - 86
"generated_note": "All figures computed by .build/precompute.py from " - 87
"gpt-tokenizer's reference (CJS) build, verified against " - 88
"the shipped browser bundle by .build/verify.py.", - 89
"encodings": { - 90
"o200k_base": {"vocab": int(o200k.ctx.eval("M.vocabularySize")), - 91
"used_by": "GPT-4o and o-series models"}, - 92
"cl100k_base": {"vocab": int(cl100k.ctx.eval("M.vocabularySize")), - 93
"used_by": "GPT-4 and GPT-3.5-turbo"}, - 94
}, - 95
"counting": [letter_story(o200k, w, l) for w, l in - 96
zip(COUNTING_WORDS, ["r", "r", "k", "s", "s"])], - 97
"numbers": [{"text": n, "tokens": o200k.tokens(n)} for n in NUMBERS], - 98
"whitespace": [{"text": w, "tokens": o200k.tokens(w)} for w in WHITESPACE], - 99
"code": {"text": CODE, "tokens": o200k.tokens(CODE)}, - 100
"prose": {"text": PROSE, "tokens": o200k.tokens(PROSE)}, - 101
"languages": [], - 102
} - 103
- 104
english_tokens = None - 105
for name, code, script, text in UDHR: - 106
n_o = o200k.count(text) - 107
n_c = cl100k.count(text) - 108
if english_tokens is None: - 109
english_tokens = n_o - 110
data["languages"].append({ - 111
"name": name, "code": code, "script": script, "text": text, - 112
"chars": len(text), - 113
"o200k": n_o, - 114
"cl100k": n_c, - 115
"chars_per_token": round(len(text) / n_o, 2), - 116
"vs_english": round(n_o / english_tokens, 2), - 117
}) - 118
- 119
# Same text, two model generations. Newer vocabulary, fewer tokens for the - 120
# same meaning — most dramatically outside English. - 121
data["encoding_shift"] = [ - 122
{"label": name, "text": text, - 123
"cl100k": cl100k.count(text), "o200k": o200k.count(text)} - 124
for name, text in [ - 125
("English prose", PROSE), - 126
("Python", CODE), - 127
("Japanese", UDHR[7][3]), - 128
("Hindi", UDHR[10][3]), - 129
("Arabic", UDHR[9][3]), - 130
] - 131
] - 132
- 133
with open(OUT, "w", encoding="utf-8") as fh: - 134
json.dump(data, fh, ensure_ascii=False, indent=1) - 135
- 136
print(f"wrote {OUT}") - 137
print(f" o200k_base vocab: {data['encodings']['o200k_base']['vocab']:,}") - 138
print(f" cl100k_base vocab: {data['encodings']['cl100k_base']['vocab']:,}") - 139
print("\n language chars o200k cl100k chars/tok vs EN") - 140
for r in data["languages"]: - 141
print(f" {r['name']:<16}{r['chars']:>6}{r['o200k']:>7}{r['cl100k']:>8}" - 142
f"{r['chars_per_token']:>11}{r['vs_english']:>7}x") - 143
- 144
- 145
if __name__ == "__main__": - 146
main()