sweedworks

← all sources

cjsload.py

Loads the tokenizer's CommonJS build under QuickJS

88 lines. This is the file the build actually runs, copied verbatim at build time.

  1. 1"""Load gpt-tokenizer's CommonJS build under QuickJS via a minimal require()
  2. 2shim. Used to independently verify the vendored dist/ bundles — if two
  3. 3different code paths agree on token IDs, the vendored bundle is trustworthy."""
  4. 4 
  5. 5import json
  6. 6import os
  7. 7import sys
  8. 8 
  9. 9HERE = os.path.dirname(os.path.abspath(__file__))
  10. 10sys.path.insert(0, os.path.join(HERE, "pylib"))
  11. 11import quickjs # noqa: E402
  12. 12 
  13. 13from tokenlib import POLYFILL # noqa: E402
  14. 14 
  15. 15CJS = os.path.join(HERE, "gpt-tokenizer-cjs")
  16. 16 
  17. 17REQUIRE_SHIM = r"""
  18. 18var __cache = {};
  19. 19function __dirname_of(p){ return p.substring(0, p.lastIndexOf('/')); }
  20. 20function __normalize(p){
  21. 21 var parts = p.split('/'), out = [];
  22. 22 for (var i = 0; i < parts.length; i++) {
  23. 23 if (parts[i] === '.' || parts[i] === '') continue;
  24. 24 if (parts[i] === '..') out.pop(); else out.push(parts[i]);
  25. 25 }
  26. 26 return '/' + out.join('/');
  27. 27}
  28. 28function __resolve(base, req){
  29. 29 var p = req.charAt(0) === '.' ? __normalize(base + '/' + req) : __normalize(req);
  30. 30 var cands = [p, p + '.js', p + '/index.js'];
  31. 31 for (var i = 0; i < cands.length; i++) if (__exists(cands[i])) return cands[i];
  32. 32 throw new Error('cannot resolve ' + req + ' from ' + base);
  33. 33}
  34. 34function __require(base, req){
  35. 35 var path = __resolve(base, req);
  36. 36 if (__cache[path]) return __cache[path].exports;
  37. 37 var mod = { exports: {} };
  38. 38 __cache[path] = mod;
  39. 39 var src = __readFile(path);
  40. 40 var dir = __dirname_of(path);
  41. 41 var fn = new Function('exports', 'require', 'module', '__filename', '__dirname', src);
  42. 42 fn(mod.exports, function(r){ return __require(dir, r); }, mod, path, dir);
  43. 43 return mod.exports;
  44. 44}
  45. 45"""
  46. 46 
  47. 47 
  48. 48class Reference:
  49. 49 """The trustworthy path: gpt-tokenizer's own CJS build, loaded from source."""
  50. 50 
  51. 51 def __init__(self, encoding="o200k_base"):
  52. 52 self.encoding = encoding
  53. 53 ctx = quickjs.Context()
  54. 54 ctx.set_memory_limit(1 << 30)
  55. 55 ctx.set_max_stack_size(1 << 22)
  56. 56 ctx.add_callable("__readFile", lambda p: open(p, encoding="utf-8").read())
  57. 57 ctx.add_callable("__exists", lambda p: os.path.isfile(p))
  58. 58 ctx.eval(POLYFILL)
  59. 59 ctx.eval(REQUIRE_SHIM)
  60. 60 ctx.set("_cjs", CJS)
  61. 61 ctx.set("_enc", encoding)
  62. 62 ctx.eval("var M = __require(_cjs, './encoding/' + _enc + '.js');")
  63. 63 ctx.eval("""
  64. 64 function tokenize(s){
  65. 65 var ids = M.encode(s), out = [];
  66. 66 for (var i = 0; i < ids.length; i++) out.push([ids[i], M.decode([ids[i]])]);
  67. 67 return JSON.stringify(out);
  68. 68 }
  69. 69 """)
  70. 70 self.ctx = ctx
  71. 71 
  72. 72 def ids(self, text):
  73. 73 self.ctx.set("_s", text)
  74. 74 return json.loads(self.ctx.eval("JSON.stringify(M.encode(_s))"))
  75. 75 
  76. 76 def tokens(self, text):
  77. 77 self.ctx.set("_s", text)
  78. 78 return [{"id": i, "text": t} for i, t in json.loads(self.ctx.eval("tokenize(_s)"))]
  79. 79 
  80. 80 def count(self, text):
  81. 81 return len(self.ids(text))
  82. 82 
  83. 83 
  84. 84if __name__ == "__main__":
  85. 85 for enc in ("cl100k_base", "o200k_base"):
  86. 86 r = Reference(enc)
  87. 87 print(f" {enc}: 'hello world' -> {r.ids('hello world')}")