tokenlib.py
Loads the shipped browser bundle under QuickJS
82 lines. This is the file the build actually runs, copied verbatim at build time.
- 1
"""Run gpt-tokenizer's browser bundle under QuickJS so the build step and the - 2
browser agree exactly on token boundaries. QuickJS has no TextDecoder/TextEncoder, - 3
so we polyfill both before loading the bundle.""" - 4
- 5
import json - 6
import os - 7
import sys - 8
- 9
HERE = os.path.dirname(os.path.abspath(__file__)) - 10
VENDOR = os.path.join(HERE, "..", "vendor") - 11
sys.path.insert(0, os.path.join(HERE, "pylib")) - 12
- 13
import quickjs # noqa: E402 - 14
- 15
POLYFILL = r""" - 16
var globalThis = this; - 17
function TextDecoder(enc){} - 18
TextDecoder.prototype.decode = function(bytes){ - 19
var out = "", i = 0, n = bytes.length; - 20
while (i < n) { - 21
var c = bytes[i++]; - 22
if (c < 0x80) out += String.fromCharCode(c); - 23
else if (c < 0xE0) out += String.fromCharCode(((c & 0x1F) << 6) | (bytes[i++] & 0x3F)); - 24
else if (c < 0xF0) out += String.fromCharCode(((c & 0x0F) << 12) | ((bytes[i++] & 0x3F) << 6) | (bytes[i++] & 0x3F)); - 25
else { - 26
var cp = ((c & 0x07) << 18) | ((bytes[i++] & 0x3F) << 12) | ((bytes[i++] & 0x3F) << 6) | (bytes[i++] & 0x3F); - 27
cp -= 0x10000; - 28
out += String.fromCharCode(0xD800 + (cp >> 10), 0xDC00 + (cp & 0x3FF)); - 29
} - 30
} - 31
return out; - 32
}; - 33
function TextEncoder(){} - 34
TextEncoder.prototype.encode = function(str){ - 35
var out = [], i = 0; - 36
while (i < str.length) { - 37
var cp = str.codePointAt(i); - 38
i += cp > 0xFFFF ? 2 : 1; - 39
if (cp < 0x80) out.push(cp); - 40
else if (cp < 0x800) out.push(0xC0 | (cp >> 6), 0x80 | (cp & 0x3F)); - 41
else if (cp < 0x10000) out.push(0xE0 | (cp >> 12), 0x80 | ((cp >> 6) & 0x3F), 0x80 | (cp & 0x3F)); - 42
else out.push(0xF0 | (cp >> 18), 0x80 | ((cp >> 12) & 0x3F), 0x80 | ((cp >> 6) & 0x3F), 0x80 | (cp & 0x3F)); - 43
} - 44
return new Uint8Array(out); - 45
}; - 46
""" - 47
- 48
- 49
class Tokenizer: - 50
def __init__(self, encoding="o200k_base"): - 51
self.encoding = encoding - 52
self.ctx = quickjs.Context() - 53
self.ctx.set_memory_limit(1 << 30) - 54
self.ctx.set_max_stack_size(1 << 22) - 55
self.ctx.eval(POLYFILL) - 56
path = os.path.join(VENDOR, "gpt-tokenizer", f"{encoding}.js") - 57
with open(path, encoding="utf-8") as fh: - 58
self.ctx.eval(fh.read()) - 59
self.ctx.eval(f"var T = globalThis.GPTTokenizer_{encoding};") - 60
self.ctx.eval(""" - 61
function tokenize(s){ - 62
var ids = T.encode(s); - 63
var out = []; - 64
for (var i = 0; i < ids.length; i++) out.push([ids[i], T.decode([ids[i]])]); - 65
return JSON.stringify(out); - 66
} - 67
""") - 68
- 69
def tokens(self, text): - 70
"""-> list of {id, text} in order.""" - 71
self.ctx.set("_input", text) - 72
raw = self.ctx.eval("tokenize(_input)") - 73
return [{"id": i, "text": t} for i, t in json.loads(raw)] - 74
- 75
def count(self, text): - 76
self.ctx.set("_input", text) - 77
return int(self.ctx.eval("T.encode(_input).length")) - 78
- 79
@property - 80
def vocab_size(self): - 81
return int(self.ctx.eval("T.vocabularySize"))