sweedworks

← all sources

tokenlib.py

Loads the shipped browser bundle under QuickJS

82 lines. This is the file the build actually runs, copied verbatim at build time.

  1. 1"""Run gpt-tokenizer's browser bundle under QuickJS so the build step and the
  2. 2browser agree exactly on token boundaries. QuickJS has no TextDecoder/TextEncoder,
  3. 3so we polyfill both before loading the bundle."""
  4. 4 
  5. 5import json
  6. 6import os
  7. 7import sys
  8. 8 
  9. 9HERE = os.path.dirname(os.path.abspath(__file__))
  10. 10VENDOR = os.path.join(HERE, "..", "vendor")
  11. 11sys.path.insert(0, os.path.join(HERE, "pylib"))
  12. 12 
  13. 13import quickjs # noqa: E402
  14. 14 
  15. 15POLYFILL = r"""
  16. 16var globalThis = this;
  17. 17function TextDecoder(enc){}
  18. 18TextDecoder.prototype.decode = function(bytes){
  19. 19 var out = "", i = 0, n = bytes.length;
  20. 20 while (i < n) {
  21. 21 var c = bytes[i++];
  22. 22 if (c < 0x80) out += String.fromCharCode(c);
  23. 23 else if (c < 0xE0) out += String.fromCharCode(((c & 0x1F) << 6) | (bytes[i++] & 0x3F));
  24. 24 else if (c < 0xF0) out += String.fromCharCode(((c & 0x0F) << 12) | ((bytes[i++] & 0x3F) << 6) | (bytes[i++] & 0x3F));
  25. 25 else {
  26. 26 var cp = ((c & 0x07) << 18) | ((bytes[i++] & 0x3F) << 12) | ((bytes[i++] & 0x3F) << 6) | (bytes[i++] & 0x3F);
  27. 27 cp -= 0x10000;
  28. 28 out += String.fromCharCode(0xD800 + (cp >> 10), 0xDC00 + (cp & 0x3FF));
  29. 29 }
  30. 30 }
  31. 31 return out;
  32. 32};
  33. 33function TextEncoder(){}
  34. 34TextEncoder.prototype.encode = function(str){
  35. 35 var out = [], i = 0;
  36. 36 while (i < str.length) {
  37. 37 var cp = str.codePointAt(i);
  38. 38 i += cp > 0xFFFF ? 2 : 1;
  39. 39 if (cp < 0x80) out.push(cp);
  40. 40 else if (cp < 0x800) out.push(0xC0 | (cp >> 6), 0x80 | (cp & 0x3F));
  41. 41 else if (cp < 0x10000) out.push(0xE0 | (cp >> 12), 0x80 | ((cp >> 6) & 0x3F), 0x80 | (cp & 0x3F));
  42. 42 else out.push(0xF0 | (cp >> 18), 0x80 | ((cp >> 12) & 0x3F), 0x80 | ((cp >> 6) & 0x3F), 0x80 | (cp & 0x3F));
  43. 43 }
  44. 44 return new Uint8Array(out);
  45. 45};
  46. 46"""
  47. 47 
  48. 48 
  49. 49class Tokenizer:
  50. 50 def __init__(self, encoding="o200k_base"):
  51. 51 self.encoding = encoding
  52. 52 self.ctx = quickjs.Context()
  53. 53 self.ctx.set_memory_limit(1 << 30)
  54. 54 self.ctx.set_max_stack_size(1 << 22)
  55. 55 self.ctx.eval(POLYFILL)
  56. 56 path = os.path.join(VENDOR, "gpt-tokenizer", f"{encoding}.js")
  57. 57 with open(path, encoding="utf-8") as fh:
  58. 58 self.ctx.eval(fh.read())
  59. 59 self.ctx.eval(f"var T = globalThis.GPTTokenizer_{encoding};")
  60. 60 self.ctx.eval("""
  61. 61 function tokenize(s){
  62. 62 var ids = T.encode(s);
  63. 63 var out = [];
  64. 64 for (var i = 0; i < ids.length; i++) out.push([ids[i], T.decode([ids[i]])]);
  65. 65 return JSON.stringify(out);
  66. 66 }
  67. 67 """)
  68. 68 
  69. 69 def tokens(self, text):
  70. 70 """-> list of {id, text} in order."""
  71. 71 self.ctx.set("_input", text)
  72. 72 raw = self.ctx.eval("tokenize(_input)")
  73. 73 return [{"id": i, "text": t} for i, t in json.loads(raw)]
  74. 74 
  75. 75 def count(self, text):
  76. 76 self.ctx.set("_input", text)
  77. 77 return int(self.ctx.eval("T.encode(_input).length"))
  78. 78 
  79. 79 @property
  80. 80 def vocab_size(self):
  81. 81 return int(self.ctx.eval("T.vocabularySize"))