sweedworks

← all sources

makebundle.py

Builds the cl100k browser bundle upstream got wrong

159 lines. This is the file the build actually runs, copied verbatim at build time.

  1. 1"""Build a browser bundle for an encoding from gpt-tokenizer's CommonJS source.
  2. 2 
  3. 3Exists because upstream's dist/cl100k_base.js is mislabeled — it emits o200k
  4. 4tokens. Rather than drop the encoding, we bundle the correct source ourselves.
  5. 5 
  6. 6The module list is not guessed: we load the encoding through the require() shim
  7. 7and take exactly the modules it actually pulled in. The result is verified
  8. 8against the reference before it is written.
  9. 9"""
  10. 10 
  11. 11import json
  12. 12import os
  13. 13import sys
  14. 14 
  15. 15HERE = os.path.dirname(os.path.abspath(__file__))
  16. 16sys.path.insert(0, os.path.join(HERE, "pylib"))
  17. 17ROOT = os.path.abspath(os.path.join(HERE, ".."))
  18. 18 
  19. 19import quickjs # noqa: E402
  20. 20 
  21. 21from cjsload import CJS, Reference # noqa: E402
  22. 22from tokenlib import POLYFILL # noqa: E402
  23. 23 
  24. 24OUT_DIR = os.path.join(ROOT, "vendor", "gpt-tokenizer")
  25. 25 
  26. 26LOADER_HEAD = """\
  27. 27/* gpt-tokenizer %(enc)s — bundled from the package's CommonJS source (MIT).
  28. 28 *
  29. 29 * Built by sweedworks.com's own build step, not by upstream. Upstream's
  30. 30 * prebuilt dist/%(enc)s.js in v3.4.0 is mislabeled: it declares this global
  31. 31 * but emits o200k_base token IDs. This bundle is checked against the reference
  32. 32 * implementation over a corpus before shipping. See /.build/makebundle.py.
  33. 33 *
  34. 34 * Source: https://github.com/niieani/gpt-tokenizer (MIT)
  35. 35 */
  36. 36(function (global) {
  37. 37 "use strict";
  38. 38 var __mods = {};
  39. 39 var __cache = {};
  40. 40 
  41. 41 function normalize(p) {
  42. 42 var parts = p.split("/"), out = [];
  43. 43 for (var i = 0; i < parts.length; i++) {
  44. 44 if (parts[i] === "." || parts[i] === "") continue;
  45. 45 if (parts[i] === "..") out.pop(); else out.push(parts[i]);
  46. 46 }
  47. 47 return out.join("/");
  48. 48 }
  49. 49 
  50. 50 function resolve(base, req) {
  51. 51 var p = req.charAt(0) === "." ? normalize(base + "/" + req) : normalize(req);
  52. 52 var cands = [p, p + ".js", p + "/index.js"];
  53. 53 for (var i = 0; i < cands.length; i++) {
  54. 54 if (Object.prototype.hasOwnProperty.call(__mods, cands[i])) return cands[i];
  55. 55 }
  56. 56 throw new Error("gpt-tokenizer bundle: cannot resolve " + req + " from " + base);
  57. 57 }
  58. 58 
  59. 59 function req_(base, request) {
  60. 60 var path = resolve(base, request);
  61. 61 if (__cache[path]) return __cache[path].exports;
  62. 62 var mod = { exports: {} };
  63. 63 __cache[path] = mod;
  64. 64 var dir = path.indexOf("/") < 0 ? "" : path.substring(0, path.lastIndexOf("/"));
  65. 65 __mods[path](mod.exports, function (r) { return req_(dir, r); }, mod);
  66. 66 return mod.exports;
  67. 67 }
  68. 68 
  69. 69 function def(path, fn) { __mods[path] = fn; }
  70. 70 
  71. 71"""
  72. 72 
  73. 73LOADER_TAIL = """
  74. 74 global.GPTTokenizer_%(enc)s = req_("", "./encoding/%(enc)s.js");
  75. 75})(typeof globalThis !== "undefined" ? globalThis :
  76. 76 typeof self !== "undefined" ? self : this);
  77. 77"""
  78. 78 
  79. 79 
  80. 80def module_list(encoding):
  81. 81 """Exactly the modules the encoding pulls in, in load order."""
  82. 82 ref = Reference(encoding)
  83. 83 paths = json.loads(ref.ctx.eval("JSON.stringify(Object.keys(__cache))"))
  84. 84 return [(os.path.relpath(p, CJS).replace(os.sep, "/"), p) for p in paths]
  85. 85 
  86. 86 
  87. 87def build(encoding):
  88. 88 mods = module_list(encoding)
  89. 89 parts = [LOADER_HEAD % {"enc": encoding}]
  90. 90 for key, full in sorted(mods):
  91. 91 src = open(full, encoding="utf-8").read()
  92. 92 parts.append(f' def({json.dumps(key)}, function (exports, require, module) {{\n')
  93. 93 parts.append(src)
  94. 94 parts.append("\n });\n\n")
  95. 95 parts.append(LOADER_TAIL % {"enc": encoding})
  96. 96 return "".join(parts), len(mods)
  97. 97 
  98. 98 
  99. 99CORPUS = [
  100. 100 "strawberry", " strawberry", "hello world", "", " ", "\n\n", "\t",
  101. 101 "1234567890", "1,234,567,890", "127.0.0.1",
  102. 102 "def total(items):\n return sum(i.price for i in items)\n",
  103. 103 "こんにちは世界", "人人生而自由", "안녕하세요", "Привет, мир",
  104. 104 "مرحبا بالعالم", "नमस्ते दुनिया", "café", "🍓", "👩‍👩‍👧‍👦",
  105. 105 "a" * 300, "The tokenizer does not know what a word is.",
  106. 106]
  107. 107 
  108. 108 
  109. 109def verify(encoding, source):
  110. 110 """The bundle must agree with the reference on every string, and round-trip."""
  111. 111 ctx = quickjs.Context()
  112. 112 ctx.set_memory_limit(1 << 30)
  113. 113 ctx.set_max_stack_size(1 << 22)
  114. 114 ctx.eval(POLYFILL)
  115. 115 ctx.eval(source)
  116. 116 ctx.eval(f"var B = globalThis.GPTTokenizer_{encoding};")
  117. 117 if ctx.eval("typeof B.encode") != "function":
  118. 118 return ["bundle exposes no encode()"]
  119. 119 
  120. 120 ref = Reference(encoding)
  121. 121 problems = []
  122. 122 for text in CORPUS:
  123. 123 ctx.set("_s", text)
  124. 124 got = json.loads(ctx.eval("JSON.stringify(B.encode(_s))"))
  125. 125 want = ref.ids(text)
  126. 126 if got != want:
  127. 127 problems.append(f"{text!r}: bundle {got[:8]} != reference {want[:8]}")
  128. 128 continue
  129. 129 if ctx.eval("B.decode(B.encode(_s))") != text:
  130. 130 problems.append(f"{text!r}: round trip changed the text")
  131. 131 return problems
  132. 132 
  133. 133 
  134. 134def main():
  135. 135 encoding = sys.argv[1] if len(sys.argv) > 1 else "cl100k_base"
  136. 136 source, n = build(encoding)
  137. 137 
  138. 138 problems = verify(encoding, source)
  139. 139 if problems:
  140. 140 print(f"FAIL {encoding}: bundle does not match reference")
  141. 141 for p in problems[:10]:
  142. 142 print(f" {p}")
  143. 143 return 1
  144. 144 
  145. 145 out = os.path.join(OUT_DIR, f"{encoding}.js")
  146. 146 with open(out, "w", encoding="utf-8") as fh:
  147. 147 fh.write(source)
  148. 148 
  149. 149 import gzip
  150. 150 gz = len(gzip.compress(source.encode()))
  151. 151 print(f"PASS {encoding}: {n} modules, matches reference on {len(CORPUS)} strings")
  152. 152 print(f" wrote vendor/gpt-tokenizer/{encoding}.js "
  153. 153 f"({len(source) / 1e6:.2f} MB raw, {gz / 1024:.0f} KB gzipped)")
  154. 154 return 0
  155. 155 
  156. 156 
  157. 157if __name__ == "__main__":
  158. 158 sys.exit(main())