verify.py
Checks both shipped tokenizer bundles against the reference
103 lines. This is the file the build actually runs, copied verbatim at build time.
- 1
"""Assert the browser bundle we ship agrees with gpt-tokenizer's reference build. - 2
- 3
This exists because gpt-tokenizer 3.4.0's dist/cl100k_base.js is mislabeled: it - 4
declares GPTTokenizer_cl100k_base but emits o200k token IDs. A filename is not - 5
evidence. Run this before every deploy. - 6
""" - 7
- 8
import sys - 9
import unicodedata - 10
- 11
from cjsload import Reference - 12
from tokenlib import Tokenizer - 13
- 14
CORPUS = [ - 15
"strawberry", " strawberry", "strawberry ", "Strawberry", "STRAWBERRY", - 16
"hello world", "Hello", " Hello", "", " ", " ", "\n", "\n\n", "\t", - 17
"1234567890", "1,234,567,890", "3.14159265", "127.0.0.1", "2024", "20240811", - 18
"def total(items):\n return sum(i.price for i in items)\n", - 19
"SELECT * FROM users WHERE id = 42;", - 20
"https://sweedworks.com/tokens/", - 21
"こんにちは世界", "の", "人人生而自由", "안녕하세요", "Привет, мир", - 22
"مرحبا بالعالم", "नमस्ते दुनिया", "Ω≈ç√∫˜µ≤≥÷", - 23
"café", "café", # precomposed vs combining — different tokens - 24
"🍓", "👩👩👧👦", "é́́", - 25
"a" * 200, "🍓" * 40, - 26
"The tokenizer does not know what a word is.", - 27
] - 28
- 29
- 30
def check_encoding(enc): - 31
"""Every bundle we serve must agree with the reference, token for token.""" - 32
shipped = Tokenizer(enc) # the bundle browsers actually download - 33
reference = Reference(enc) # the package's own CJS source - 34
- 35
failures = [] - 36
for text in CORPUS: - 37
a = [t["id"] for t in shipped.tokens(text)] - 38
b = reference.ids(text) - 39
if a != b: - 40
failures.append((text, a, b)) - 41
- 42
if failures: - 43
print(f"FAIL {enc}: {len(failures)} of {len(CORPUS)} strings mismatch") - 44
for text, a, b in failures[:6]: - 45
print(f" {text!r}\n shipped {a[:8]}\n reference {b[:8]}") - 46
return 1 - 47
print(f"PASS {enc}: {len(CORPUS)} strings — shipped bundle matches reference") - 48
- 49
bad = [] - 50
for text in CORPUS: - 51
shipped.ctx.set("_s", text) - 52
if shipped.ctx.eval("T.decode(T.encode(_s))") != text: - 53
bad.append(text) - 54
if bad: - 55
print(f"FAIL {enc}: round trip changed {len(bad)} string(s): {bad[:3]}") - 56
return 1 - 57
print(f"PASS {enc}: round trip decode(encode(x)) == x") - 58
return 0 - 59
- 60
- 61
def main(): - 62
bad = 0 - 63
for enc in ("o200k_base", "cl100k_base"): - 64
bad += check_encoding(enc) - 65
if bad: - 66
return 1 - 67
- 68
# The two encodings must not be the same thing wearing different names. - 69
# This is the exact check that caught upstream's mislabeled bundle. - 70
a = Tokenizer("o200k_base") - 71
b = Tokenizer("cl100k_base") - 72
if a.count("hello world") == b.count("hello world") and \ - 73
[t["id"] for t in a.tokens("hello world")] == \ - 74
[t["id"] for t in b.tokens("hello world")]: - 75
print("FAIL the two shipped bundles produce identical IDs — " - 76
"one of them is mislabeled") - 77
return 1 - 78
print("PASS the two shipped bundles are genuinely different vocabularies") - 79
- 80
shipped = a - 81
reference = Reference("o200k_base") - 82
- 83
# Sanity: this encoding must actually be o200k, not something wearing its name. - 84
# "hello world" is [15339,1917] under cl100k and [24912,2375] under o200k, so - 85
# it distinguishes the two. The strawberry split is asserted on decoded text - 86
# rather than raw ids — that's the claim the site actually makes. - 87
if reference.ids("hello world") != [24912, 2375]: - 88
print(f"FAIL identity: 'hello world' -> {reference.ids('hello world')}") - 89
return 1 - 90
split = [t["text"] for t in shipped.tokens("strawberry")] - 91
if split != ["st", "raw", "berry"]: - 92
print(f"FAIL identity: 'strawberry' splits as {split}, expected st/raw/berry") - 93
return 1 - 94
if shipped.count(" strawberry") != 1: - 95
print("FAIL identity: ' strawberry' should be a single token") - 96
return 1 - 97
print("PASS identity — encoding is genuinely o200k_base") - 98
return 0 - 99
- 100
- 101
if __name__ == "__main__": - 102
sys.exit(main())