sweedworks

← all sources

precompute.py

Computes the figures on /tokens/

147 lines. This is the file the build actually runs, copied verbatim at build time.

  1. 1"""Generate tokens/data.json — every number shown on the page is computed here,
  2. 2never hand-written, so the prose can't drift from the tokenizer."""
  3. 3 
  4. 4import json
  5. 5import os
  6. 6 
  7. 7from cjsload import Reference
  8. 8 
  9. 9OUT = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "tokens", "data.json")
  10. 10 
  11. 11# Article 1 of the Universal Declaration of Human Rights, in the UN's official
  12. 12# translations. Chosen because it's the same meaning in every row — which is the
  13. 13# whole point: identical content, wildly different token cost.
  14. 14UDHR = [
  15. 15 ("English", "en", "Latin",
  16. 16 "All human beings are born free and equal in dignity and rights. They are "
  17. 17 "endowed with reason and conscience and should act towards one another in a "
  18. 18 "spirit of brotherhood."),
  19. 19 ("Spanish", "es", "Latin",
  20. 20 "Todos los seres humanos nacen libres e iguales en dignidad y derechos y, "
  21. 21 "dotados como están de razón y conciencia, deben comportarse fraternalmente "
  22. 22 "los unos con los otros."),
  23. 23 ("French", "fr", "Latin",
  24. 24 "Tous les êtres humains naissent libres et égaux en dignité et en droits. "
  25. 25 "Ils sont doués de raison et de conscience et doivent agir les uns envers "
  26. 26 "les autres dans un esprit de fraternité."),
  27. 27 ("German", "de", "Latin",
  28. 28 "Alle Menschen sind frei und gleich an Würde und Rechten geboren. Sie sind "
  29. 29 "mit Vernunft und Gewissen begabt und sollen einander im Geiste der "
  30. 30 "Brüderlichkeit begegnen."),
  31. 31 ("Portuguese", "pt", "Latin",
  32. 32 "Todos os seres humanos nascem livres e iguais em dignidade e direitos. São "
  33. 33 "dotados de razão e consciência e devem agir uns para com os outros em "
  34. 34 "espírito de fraternidade."),
  35. 35 ("Russian", "ru", "Cyrillic",
  36. 36 "Все люди рождаются свободными и равными в своем достоинстве и правах. Они "
  37. 37 "наделены разумом и совестью и должны поступать в отношении друг друга в "
  38. 38 "духе братства."),
  39. 39 ("Chinese", "zh", "Han",
  40. 40 "人人生而自由,在尊严和权利上一律平等。他们赋有理性和良心,并应以兄弟关系的精神相对待。"),
  41. 41 ("Japanese", "ja", "Japanese",
  42. 42 "すべての人間は、生れながらにして自由であり、かつ、尊厳と権利とについて平等である。"
  43. 43 "人間は、理性と良心とを授けられており、互いに同胞の精神をもって行動しなければならない。"),
  44. 44 ("Korean", "ko", "Hangul",
  45. 45 "모든 인간은 태어날 때부터 자유로우며 그 존엄과 권리에 있어 동등하다. 인간은 천부적으로 "
  46. 46 "이성과 양심을 부여받았으며 서로 형제애의 정신으로 행동하여야 한다."),
  47. 47 ("Arabic", "ar", "Arabic",
  48. 48 "يولد جميع الناس أحراراً متساوين في الكرامة والحقوق. وقد وهبوا عقلاً وضميراً "
  49. 49 "وعليهم أن يعامل بعضهم بعضاً بروح الإخاء."),
  50. 50 ("Hindi", "hi", "Devanagari",
  51. 51 "सभी मनुष्यों को गौरव और अधिकारों के मामले में जन्मजात स्वतन्त्रता और समानता प्राप्त है। "
  52. 52 "उन्हें बुद्धि और अन्तरात्मा की देन प्राप्त है और परस्पर उन्हें भाईचारे के भाव से बर्ताव करना चाहिए।"),
  53. 53]
  54. 54 
  55. 55COUNTING_WORDS = ["strawberry", "raspberry", "bookkeeper", "Mississippi", "unsuccessfully"]
  56. 56 
  57. 57NUMBERS = ["1234567890", "1,234,567,890", "3.14159265", "2024", "20240811", "127.0.0.1"]
  58. 58 
  59. 59WHITESPACE = ["strawberry", " strawberry", "strawberry ", "Strawberry", "STRAWBERRY", " strawberry"]
  60. 60 
  61. 61CODE = 'def total(items):\n return sum(i.price for i in items)\n'
  62. 62 
  63. 63PROSE = (
  64. 64 "The tokenizer does not know what a word is. It knows which byte sequences "
  65. 65 "showed up together often enough during training to deserve their own number."
  66. 66)
  67. 67 
  68. 68 
  69. 69def letter_story(tok, word, letter):
  70. 70 """How the model sees a word it's being asked to spell."""
  71. 71 toks = tok.tokens(word)
  72. 72 return {
  73. 73 "word": word,
  74. 74 "letter": letter,
  75. 75 "true_count": word.lower().count(letter),
  76. 76 "tokens": toks,
  77. 77 "token_count": len(toks),
  78. 78 }
  79. 79 
  80. 80 
  81. 81def main():
  82. 82 o200k = Reference("o200k_base")
  83. 83 cl100k = Reference("cl100k_base")
  84. 84 
  85. 85 data = {
  86. 86 "generated_note": "All figures computed by .build/precompute.py from "
  87. 87 "gpt-tokenizer's reference (CJS) build, verified against "
  88. 88 "the shipped browser bundle by .build/verify.py.",
  89. 89 "encodings": {
  90. 90 "o200k_base": {"vocab": int(o200k.ctx.eval("M.vocabularySize")),
  91. 91 "used_by": "GPT-4o and o-series models"},
  92. 92 "cl100k_base": {"vocab": int(cl100k.ctx.eval("M.vocabularySize")),
  93. 93 "used_by": "GPT-4 and GPT-3.5-turbo"},
  94. 94 },
  95. 95 "counting": [letter_story(o200k, w, l) for w, l in
  96. 96 zip(COUNTING_WORDS, ["r", "r", "k", "s", "s"])],
  97. 97 "numbers": [{"text": n, "tokens": o200k.tokens(n)} for n in NUMBERS],
  98. 98 "whitespace": [{"text": w, "tokens": o200k.tokens(w)} for w in WHITESPACE],
  99. 99 "code": {"text": CODE, "tokens": o200k.tokens(CODE)},
  100. 100 "prose": {"text": PROSE, "tokens": o200k.tokens(PROSE)},
  101. 101 "languages": [],
  102. 102 }
  103. 103 
  104. 104 english_tokens = None
  105. 105 for name, code, script, text in UDHR:
  106. 106 n_o = o200k.count(text)
  107. 107 n_c = cl100k.count(text)
  108. 108 if english_tokens is None:
  109. 109 english_tokens = n_o
  110. 110 data["languages"].append({
  111. 111 "name": name, "code": code, "script": script, "text": text,
  112. 112 "chars": len(text),
  113. 113 "o200k": n_o,
  114. 114 "cl100k": n_c,
  115. 115 "chars_per_token": round(len(text) / n_o, 2),
  116. 116 "vs_english": round(n_o / english_tokens, 2),
  117. 117 })
  118. 118 
  119. 119 # Same text, two model generations. Newer vocabulary, fewer tokens for the
  120. 120 # same meaning — most dramatically outside English.
  121. 121 data["encoding_shift"] = [
  122. 122 {"label": name, "text": text,
  123. 123 "cl100k": cl100k.count(text), "o200k": o200k.count(text)}
  124. 124 for name, text in [
  125. 125 ("English prose", PROSE),
  126. 126 ("Python", CODE),
  127. 127 ("Japanese", UDHR[7][3]),
  128. 128 ("Hindi", UDHR[10][3]),
  129. 129 ("Arabic", UDHR[9][3]),
  130. 130 ]
  131. 131 ]
  132. 132 
  133. 133 with open(OUT, "w", encoding="utf-8") as fh:
  134. 134 json.dump(data, fh, ensure_ascii=False, indent=1)
  135. 135 
  136. 136 print(f"wrote {OUT}")
  137. 137 print(f" o200k_base vocab: {data['encodings']['o200k_base']['vocab']:,}")
  138. 138 print(f" cl100k_base vocab: {data['encodings']['cl100k_base']['vocab']:,}")
  139. 139 print("\n language chars o200k cl100k chars/tok vs EN")
  140. 140 for r in data["languages"]:
  141. 141 print(f" {r['name']:<16}{r['chars']:>6}{r['o200k']:>7}{r['cl100k']:>8}"
  142. 142 f"{r['chars_per_token']:>11}{r['vs_english']:>7}x")
  143. 143 
  144. 144 
  145. 145if __name__ == "__main__":
  146. 146 main()