cjsload.py
Loads the tokenizer's CommonJS build under QuickJS
88 lines. This is the file the build actually runs, copied verbatim at build time.
- 1
"""Load gpt-tokenizer's CommonJS build under QuickJS via a minimal require() - 2
shim. Used to independently verify the vendored dist/ bundles — if two - 3
different code paths agree on token IDs, the vendored bundle is trustworthy.""" - 4
- 5
import json - 6
import os - 7
import sys - 8
- 9
HERE = os.path.dirname(os.path.abspath(__file__)) - 10
sys.path.insert(0, os.path.join(HERE, "pylib")) - 11
import quickjs # noqa: E402 - 12
- 13
from tokenlib import POLYFILL # noqa: E402 - 14
- 15
CJS = os.path.join(HERE, "gpt-tokenizer-cjs") - 16
- 17
REQUIRE_SHIM = r""" - 18
var __cache = {}; - 19
function __dirname_of(p){ return p.substring(0, p.lastIndexOf('/')); } - 20
function __normalize(p){ - 21
var parts = p.split('/'), out = []; - 22
for (var i = 0; i < parts.length; i++) { - 23
if (parts[i] === '.' || parts[i] === '') continue; - 24
if (parts[i] === '..') out.pop(); else out.push(parts[i]); - 25
} - 26
return '/' + out.join('/'); - 27
} - 28
function __resolve(base, req){ - 29
var p = req.charAt(0) === '.' ? __normalize(base + '/' + req) : __normalize(req); - 30
var cands = [p, p + '.js', p + '/index.js']; - 31
for (var i = 0; i < cands.length; i++) if (__exists(cands[i])) return cands[i]; - 32
throw new Error('cannot resolve ' + req + ' from ' + base); - 33
} - 34
function __require(base, req){ - 35
var path = __resolve(base, req); - 36
if (__cache[path]) return __cache[path].exports; - 37
var mod = { exports: {} }; - 38
__cache[path] = mod; - 39
var src = __readFile(path); - 40
var dir = __dirname_of(path); - 41
var fn = new Function('exports', 'require', 'module', '__filename', '__dirname', src); - 42
fn(mod.exports, function(r){ return __require(dir, r); }, mod, path, dir); - 43
return mod.exports; - 44
} - 45
""" - 46
- 47
- 48
class Reference: - 49
"""The trustworthy path: gpt-tokenizer's own CJS build, loaded from source.""" - 50
- 51
def __init__(self, encoding="o200k_base"): - 52
self.encoding = encoding - 53
ctx = quickjs.Context() - 54
ctx.set_memory_limit(1 << 30) - 55
ctx.set_max_stack_size(1 << 22) - 56
ctx.add_callable("__readFile", lambda p: open(p, encoding="utf-8").read()) - 57
ctx.add_callable("__exists", lambda p: os.path.isfile(p)) - 58
ctx.eval(POLYFILL) - 59
ctx.eval(REQUIRE_SHIM) - 60
ctx.set("_cjs", CJS) - 61
ctx.set("_enc", encoding) - 62
ctx.eval("var M = __require(_cjs, './encoding/' + _enc + '.js');") - 63
ctx.eval(""" - 64
function tokenize(s){ - 65
var ids = M.encode(s), out = []; - 66
for (var i = 0; i < ids.length; i++) out.push([ids[i], M.decode([ids[i]])]); - 67
return JSON.stringify(out); - 68
} - 69
""") - 70
self.ctx = ctx - 71
- 72
def ids(self, text): - 73
self.ctx.set("_s", text) - 74
return json.loads(self.ctx.eval("JSON.stringify(M.encode(_s))")) - 75
- 76
def tokens(self, text): - 77
self.ctx.set("_s", text) - 78
return [{"id": i, "text": t} for i, t in json.loads(self.ctx.eval("tokenize(_s)"))] - 79
- 80
def count(self, text): - 81
return len(self.ids(text)) - 82
- 83
- 84
if __name__ == "__main__": - 85
for enc in ("cl100k_base", "o200k_base"): - 86
r = Reference(enc) - 87
print(f" {enc}: 'hello world' -> {r.ids('hello world')}")