sweedworks

← all sources

verify.py

Checks both shipped tokenizer bundles against the reference

103 lines. This is the file the build actually runs, copied verbatim at build time.

  1. 1"""Assert the browser bundle we ship agrees with gpt-tokenizer's reference build.
  2. 2 
  3. 3This exists because gpt-tokenizer 3.4.0's dist/cl100k_base.js is mislabeled: it
  4. 4declares GPTTokenizer_cl100k_base but emits o200k token IDs. A filename is not
  5. 5evidence. Run this before every deploy.
  6. 6"""
  7. 7 
  8. 8import sys
  9. 9import unicodedata
  10. 10 
  11. 11from cjsload import Reference
  12. 12from tokenlib import Tokenizer
  13. 13 
  14. 14CORPUS = [
  15. 15 "strawberry", " strawberry", "strawberry ", "Strawberry", "STRAWBERRY",
  16. 16 "hello world", "Hello", " Hello", "", " ", " ", "\n", "\n\n", "\t",
  17. 17 "1234567890", "1,234,567,890", "3.14159265", "127.0.0.1", "2024", "20240811",
  18. 18 "def total(items):\n return sum(i.price for i in items)\n",
  19. 19 "SELECT * FROM users WHERE id = 42;",
  20. 20 "https://sweedworks.com/tokens/",
  21. 21 "こんにちは世界", "の", "人人生而自由", "안녕하세요", "Привет, мир",
  22. 22 "مرحبا بالعالم", "नमस्ते दुनिया", "Ω≈ç√∫˜µ≤≥÷",
  23. 23 "café", "café", # precomposed vs combining — different tokens
  24. 24 "🍓", "👩‍👩‍👧‍👦", "é́́",
  25. 25 "a" * 200, "🍓" * 40,
  26. 26 "The tokenizer does not know what a word is.",
  27. 27]
  28. 28 
  29. 29 
  30. 30def check_encoding(enc):
  31. 31 """Every bundle we serve must agree with the reference, token for token."""
  32. 32 shipped = Tokenizer(enc) # the bundle browsers actually download
  33. 33 reference = Reference(enc) # the package's own CJS source
  34. 34 
  35. 35 failures = []
  36. 36 for text in CORPUS:
  37. 37 a = [t["id"] for t in shipped.tokens(text)]
  38. 38 b = reference.ids(text)
  39. 39 if a != b:
  40. 40 failures.append((text, a, b))
  41. 41 
  42. 42 if failures:
  43. 43 print(f"FAIL {enc}: {len(failures)} of {len(CORPUS)} strings mismatch")
  44. 44 for text, a, b in failures[:6]:
  45. 45 print(f" {text!r}\n shipped {a[:8]}\n reference {b[:8]}")
  46. 46 return 1
  47. 47 print(f"PASS {enc}: {len(CORPUS)} strings — shipped bundle matches reference")
  48. 48 
  49. 49 bad = []
  50. 50 for text in CORPUS:
  51. 51 shipped.ctx.set("_s", text)
  52. 52 if shipped.ctx.eval("T.decode(T.encode(_s))") != text:
  53. 53 bad.append(text)
  54. 54 if bad:
  55. 55 print(f"FAIL {enc}: round trip changed {len(bad)} string(s): {bad[:3]}")
  56. 56 return 1
  57. 57 print(f"PASS {enc}: round trip decode(encode(x)) == x")
  58. 58 return 0
  59. 59 
  60. 60 
  61. 61def main():
  62. 62 bad = 0
  63. 63 for enc in ("o200k_base", "cl100k_base"):
  64. 64 bad += check_encoding(enc)
  65. 65 if bad:
  66. 66 return 1
  67. 67 
  68. 68 # The two encodings must not be the same thing wearing different names.
  69. 69 # This is the exact check that caught upstream's mislabeled bundle.
  70. 70 a = Tokenizer("o200k_base")
  71. 71 b = Tokenizer("cl100k_base")
  72. 72 if a.count("hello world") == b.count("hello world") and \
  73. 73 [t["id"] for t in a.tokens("hello world")] == \
  74. 74 [t["id"] for t in b.tokens("hello world")]:
  75. 75 print("FAIL the two shipped bundles produce identical IDs — "
  76. 76 "one of them is mislabeled")
  77. 77 return 1
  78. 78 print("PASS the two shipped bundles are genuinely different vocabularies")
  79. 79 
  80. 80 shipped = a
  81. 81 reference = Reference("o200k_base")
  82. 82 
  83. 83 # Sanity: this encoding must actually be o200k, not something wearing its name.
  84. 84 # "hello world" is [15339,1917] under cl100k and [24912,2375] under o200k, so
  85. 85 # it distinguishes the two. The strawberry split is asserted on decoded text
  86. 86 # rather than raw ids — that's the claim the site actually makes.
  87. 87 if reference.ids("hello world") != [24912, 2375]:
  88. 88 print(f"FAIL identity: 'hello world' -> {reference.ids('hello world')}")
  89. 89 return 1
  90. 90 split = [t["text"] for t in shipped.tokens("strawberry")]
  91. 91 if split != ["st", "raw", "berry"]:
  92. 92 print(f"FAIL identity: 'strawberry' splits as {split}, expected st/raw/berry")
  93. 93 return 1
  94. 94 if shipped.count(" strawberry") != 1:
  95. 95 print("FAIL identity: ' strawberry' should be a single token")
  96. 96 return 1
  97. 97 print("PASS identity — encoding is genuinely o200k_base")
  98. 98 return 0
  99. 99 
  100. 100 
  101. 101if __name__ == "__main__":
  102. 102 sys.exit(main())