makebundle.py
Builds the cl100k browser bundle upstream got wrong
159 lines. This is the file the build actually runs, copied verbatim at build time.
- 1
"""Build a browser bundle for an encoding from gpt-tokenizer's CommonJS source. - 2
- 3
Exists because upstream's dist/cl100k_base.js is mislabeled — it emits o200k - 4
tokens. Rather than drop the encoding, we bundle the correct source ourselves. - 5
- 6
The module list is not guessed: we load the encoding through the require() shim - 7
and take exactly the modules it actually pulled in. The result is verified - 8
against the reference before it is written. - 9
""" - 10
- 11
import json - 12
import os - 13
import sys - 14
- 15
HERE = os.path.dirname(os.path.abspath(__file__)) - 16
sys.path.insert(0, os.path.join(HERE, "pylib")) - 17
ROOT = os.path.abspath(os.path.join(HERE, "..")) - 18
- 19
import quickjs # noqa: E402 - 20
- 21
from cjsload import CJS, Reference # noqa: E402 - 22
from tokenlib import POLYFILL # noqa: E402 - 23
- 24
OUT_DIR = os.path.join(ROOT, "vendor", "gpt-tokenizer") - 25
- 26
LOADER_HEAD = """\ - 27
/* gpt-tokenizer %(enc)s — bundled from the package's CommonJS source (MIT). - 28
* - 29
* Built by sweedworks.com's own build step, not by upstream. Upstream's - 30
* prebuilt dist/%(enc)s.js in v3.4.0 is mislabeled: it declares this global - 31
* but emits o200k_base token IDs. This bundle is checked against the reference - 32
* implementation over a corpus before shipping. See /.build/makebundle.py. - 33
* - 34
* Source: https://github.com/niieani/gpt-tokenizer (MIT) - 35
*/ - 36
(function (global) { - 37
"use strict"; - 38
var __mods = {}; - 39
var __cache = {}; - 40
- 41
function normalize(p) { - 42
var parts = p.split("/"), out = []; - 43
for (var i = 0; i < parts.length; i++) { - 44
if (parts[i] === "." || parts[i] === "") continue; - 45
if (parts[i] === "..") out.pop(); else out.push(parts[i]); - 46
} - 47
return out.join("/"); - 48
} - 49
- 50
function resolve(base, req) { - 51
var p = req.charAt(0) === "." ? normalize(base + "/" + req) : normalize(req); - 52
var cands = [p, p + ".js", p + "/index.js"]; - 53
for (var i = 0; i < cands.length; i++) { - 54
if (Object.prototype.hasOwnProperty.call(__mods, cands[i])) return cands[i]; - 55
} - 56
throw new Error("gpt-tokenizer bundle: cannot resolve " + req + " from " + base); - 57
} - 58
- 59
function req_(base, request) { - 60
var path = resolve(base, request); - 61
if (__cache[path]) return __cache[path].exports; - 62
var mod = { exports: {} }; - 63
__cache[path] = mod; - 64
var dir = path.indexOf("/") < 0 ? "" : path.substring(0, path.lastIndexOf("/")); - 65
__mods[path](mod.exports, function (r) { return req_(dir, r); }, mod); - 66
return mod.exports; - 67
} - 68
- 69
function def(path, fn) { __mods[path] = fn; } - 70
- 71
""" - 72
- 73
LOADER_TAIL = """ - 74
global.GPTTokenizer_%(enc)s = req_("", "./encoding/%(enc)s.js"); - 75
})(typeof globalThis !== "undefined" ? globalThis : - 76
typeof self !== "undefined" ? self : this); - 77
""" - 78
- 79
- 80
def module_list(encoding): - 81
"""Exactly the modules the encoding pulls in, in load order.""" - 82
ref = Reference(encoding) - 83
paths = json.loads(ref.ctx.eval("JSON.stringify(Object.keys(__cache))")) - 84
return [(os.path.relpath(p, CJS).replace(os.sep, "/"), p) for p in paths] - 85
- 86
- 87
def build(encoding): - 88
mods = module_list(encoding) - 89
parts = [LOADER_HEAD % {"enc": encoding}] - 90
for key, full in sorted(mods): - 91
src = open(full, encoding="utf-8").read() - 92
parts.append(f' def({json.dumps(key)}, function (exports, require, module) {{\n') - 93
parts.append(src) - 94
parts.append("\n });\n\n") - 95
parts.append(LOADER_TAIL % {"enc": encoding}) - 96
return "".join(parts), len(mods) - 97
- 98
- 99
CORPUS = [ - 100
"strawberry", " strawberry", "hello world", "", " ", "\n\n", "\t", - 101
"1234567890", "1,234,567,890", "127.0.0.1", - 102
"def total(items):\n return sum(i.price for i in items)\n", - 103
"こんにちは世界", "人人生而自由", "안녕하세요", "Привет, мир", - 104
"مرحبا بالعالم", "नमस्ते दुनिया", "café", "🍓", "👩👩👧👦", - 105
"a" * 300, "The tokenizer does not know what a word is.", - 106
] - 107
- 108
- 109
def verify(encoding, source): - 110
"""The bundle must agree with the reference on every string, and round-trip.""" - 111
ctx = quickjs.Context() - 112
ctx.set_memory_limit(1 << 30) - 113
ctx.set_max_stack_size(1 << 22) - 114
ctx.eval(POLYFILL) - 115
ctx.eval(source) - 116
ctx.eval(f"var B = globalThis.GPTTokenizer_{encoding};") - 117
if ctx.eval("typeof B.encode") != "function": - 118
return ["bundle exposes no encode()"] - 119
- 120
ref = Reference(encoding) - 121
problems = [] - 122
for text in CORPUS: - 123
ctx.set("_s", text) - 124
got = json.loads(ctx.eval("JSON.stringify(B.encode(_s))")) - 125
want = ref.ids(text) - 126
if got != want: - 127
problems.append(f"{text!r}: bundle {got[:8]} != reference {want[:8]}") - 128
continue - 129
if ctx.eval("B.decode(B.encode(_s))") != text: - 130
problems.append(f"{text!r}: round trip changed the text") - 131
return problems - 132
- 133
- 134
def main(): - 135
encoding = sys.argv[1] if len(sys.argv) > 1 else "cl100k_base" - 136
source, n = build(encoding) - 137
- 138
problems = verify(encoding, source) - 139
if problems: - 140
print(f"FAIL {encoding}: bundle does not match reference") - 141
for p in problems[:10]: - 142
print(f" {p}") - 143
return 1 - 144
- 145
out = os.path.join(OUT_DIR, f"{encoding}.js") - 146
with open(out, "w", encoding="utf-8") as fh: - 147
fh.write(source) - 148
- 149
import gzip - 150
gz = len(gzip.compress(source.encode())) - 151
print(f"PASS {encoding}: {n} modules, matches reference on {len(CORPUS)} strings") - 152
print(f" wrote vendor/gpt-tokenizer/{encoding}.js " - 153
f"({len(source) / 1e6:.2f} MB raw, {gz / 1024:.0f} KB gzipped)") - 154
return 0 - 155
- 156
- 157
if __name__ == "__main__": - 158
sys.exit(main())