samcheng0 commited on
Commit
58c3c64
·
verified ·
1 Parent(s): 7778d0f

Upload folder using huggingface_hub

Browse files
Files changed (5) hide show
  1. checkpoint.pt +2 -2
  2. gen_tokenizer.py +89 -1824
  3. infer_gguf.py +299 -75
  4. model_tiny.py +121 -6
  5. tokenizer.json +0 -0
checkpoint.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8cc0d5d0927b27c8c3921d96f21029f47857b105acb3b68eed4d521ac2f20ec9
3
- size 2385856
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7341bcfb74b09c7c8c8bd1946f53f6e44823a574d9b2e9f0040707a43c6c05cd
3
+ size 3906061
gen_tokenizer.py CHANGED
@@ -1,1833 +1,98 @@
 
 
 
 
1
  import json
 
 
2
 
3
- # === GPT-2 byte-to-unicode mapping ===
4
- def bytes_to_unicode():
5
- bs = list(range(ord("!"), ord("~") + 1)) + list(range(ord("¡"), ord("¬") + 1)) + list(range(ord("®"), ord("ÿ") + 1))
6
- cs = bs[:]
7
- n = 0
8
- for b in range(256):
9
- if b not in bs:
10
- bs.append(b)
11
- cs.append(256 + n)
12
- n += 1
13
- return {b: chr(c) for b, c in zip(bs, cs)}
14
 
15
- byte2char = bytes_to_unicode()
16
- # Reverse mapping: char -> byte
17
- char2byte = {c: b for b, c in byte2char.items()}
18
 
19
- # === Vocab structure ===
20
- # IDs 0-50: 51 special tokens
21
- # IDs 51-306: 256 byte-level chars
22
- # IDs 307-4095: ~3789 curated English words + subwords
23
  special_tokens = [
24
- ("<unk>", 0),
25
- ("<s>", 1),
26
- ("</s>", 2),
27
- ("<pad>", 3),
28
- ("<|system|>", 4),
29
- ("<|user|>", 5),
30
- ("<|assistant|>", 6),
31
- ("<think>", 7),
32
- ("</think>", 8),
33
- ("[INST]", 9),
34
- ("[/INST]", 10),
35
- ("<|begin_of_thought|>", 11),
36
- ("<|end_of_thought|>", 12),
37
- ("<|reflect|>", 13),
38
- ("<|revise|>", 14),
39
- ("<|verify|>", 15),
40
- ("<|code|>", 16),
41
- ("<|text|>", 17),
42
- ("<|math|>", 18),
43
- ("<|think|>", 19),
44
- ("<|answer|>", 20),
45
- ("<|step|>", 21),
46
- ("<|reason|>", 22),
47
- ("<|check|>", 23),
48
- ("<|output|>", 24),
49
- ("<|plan|>", 25),
50
- ("<|solve|>", 26),
51
- ("<|analyze|>", 27),
52
- ("<|conclude|>", 28),
53
- ("<|approach|>", 29),
54
- ("<|alternative|>", 30),
55
- ("<|summary|>", 31),
56
- ("<|question|>", 32),
57
- ("<|hint|>", 33),
58
- ("<|example|>", 34),
59
- ("<|correct|>", 35),
60
- ("<|incorrect|>", 36),
61
- ("<|feedback|>", 37),
62
- ("<|start|>", 38),
63
- ("<|end|>", 39),
64
- ("<|sep|>", 40),
65
- ("<|cls|>", 41),
66
- ("<|tool|>", 42),
67
- ("<|function|>", 43),
68
- ("<|result|>", 44),
69
- ("<|input|>", 45),
70
- ("<|detect|>", 46),
71
- ("<|context|>", 47),
72
- ("<|proof|>", 48),
73
- ("<|lemma|>", 49),
74
- ("<|theorem|>", 50),
75
- ]
76
-
77
- # IDs 51-306: 256 byte-level characters
78
- byte_tokens = []
79
- for b in range(256):
80
- byte_tokens.append((byte2char[b], 51 + b))
81
-
82
- # IDs 307+: common words (variable count, no filler)
83
- # Most frequent English words + programming terms
84
- common_words = [
85
- "the", "be", "to", "of", "and", "a", "in", "that", "have", "I",
86
- "it", "for", "not", "on", "with", "he", "as", "you", "do", "at",
87
- "this", "but", "his", "by", "from", "they", "we", "say", "her", "she",
88
- "or", "an", "will", "my", "one", "all", "would", "there", "their", "what",
89
- "so", "up", "out", "if", "about", "who", "get", "which", "go", "me",
90
- "when", "make", "can", "like", "time", "no", "just", "him", "know", "take",
91
- "people", "into", "year", "your", "good", "some", "could", "them", "see", "other",
92
- "than", "then", "now", "look", "only", "come", "its", "over", "think", "also",
93
- "back", "after", "use", "two", "how", "our", "work", "first", "well", "way",
94
- "even", "new", "want", "because", "any", "these", "give", "day", "most", "us",
95
- "is", "was", "are", "were", "been", "has", "had", "did", "does", "am",
96
- "being", "having", "doing", "saying", "going", "getting", "making", "knowing", "taking", "thinking",
97
- "come", "coming", "came", "go", "goes", "gone", "going", "went",
98
- "see", "saw", "seen", "seeing", "say", "said", "says", "saying",
99
- "get", "got", "gotten", "getting", "make", "made", "makes", "making",
100
- "know", "knew", "known", "knows", "think", "thought", "thinks", "thinking",
101
- "take", "took", "taken", "takes", "taking", "give", "gave", "given", "gives", "giving",
102
- "find", "found", "finds", "finding", "tell", "told", "tells", "telling",
103
- "ask", "asked", "asks", "asking", "show", "showed", "shown", "shows", "showing",
104
- "try", "tried", "tries", "trying", "leave", "left", "leaves", "leaving",
105
- "call", "called", "calls", "calling", "keep", "kept", "keeps", "keeping",
106
- "let", "lets", "letting", "begin", "began", "begun", "begins", "beginning",
107
- "seem", "seemed", "seems", "seeming", "help", "helped", "helps", "helping",
108
- "turn", "turned", "turns", "turning", "start", "started", "starts", "starting",
109
- "bring", "brought", "brings", "bringing", "happen", "happened", "happens", "happening",
110
- "write", "wrote", "written", "writes", "writing", "provide", "provided", "provides", "providing",
111
- "consider", "considered", "considers", "considering", "appear", "appeared", "appears", "appearing",
112
- "follow", "followed", "follows", "following", "change", "changed", "changes", "changing",
113
- "form", "formed", "forms", "forming", "need", "needed", "needs", "needing",
114
- "set", "sets", "setting", "put", "puts", "putting", "run", "runs", "running",
115
- "move", "moved", "moves", "moving", "stand", "stood", "stands", "standing",
116
- "win", "won", "wins", "winning", "play", "played", "plays", "playing",
117
- "point", "points", "pointed", "pointing", "large", "small", "big", "little",
118
- "long", "short", "high", "low", "old", "young", "great", "important",
119
- "different", "same", "other", "many", "much", "more", "most", "few",
120
- "own", "very", "such", "still", "just", "also", "even", "too",
121
- "here", "there", "where", "when", "why", "how", "what", "which",
122
- "while", "though", "although", "until", "since", "before", "after", "during",
123
- "without", "within", "between", "through", "across", "around", "above", "below",
124
- "under", "over", "again", "ever", "never", "always", "often", "sometimes",
125
- "together", "alone", "already", "yet", "still", "almost", "quite", "rather",
126
- "well", "bad", "better", "worse", "best", "worst", "more", "less",
127
- "every", "each", "both", "either", "neither", "all", "any", "none",
128
- "thing", "things", "way", "ways", "time", "times", "year", "years",
129
- "day", "days", "week", "weeks", "month", "months", "part", "parts",
130
- "place", "places", "case", "cases", "point", "points", "world", "worlds",
131
- "number", "numbers", "group", "groups", "system", "systems", "program", "programs",
132
- "data", "information", "problem", "problems", "solution", "solutions",
133
- "method", "methods", "result", "results", "process", "processes",
134
- "function", "functions", "value", "values", "type", "types",
135
- "state", "states", "model", "models", "level", "levels",
136
- "line", "lines", "file", "files", "code", "codes",
137
- "set", "sets", "list", "lists", "array", "arrays",
138
- "object", "objects", "class", "classes", "property", "properties",
139
- "input", "inputs", "output", "outputs", "return", "returns",
140
- "define", "defined", "defines", "defining", "declare", "declared",
141
- "import", "imports", "export", "exports", "include", "includes",
142
- "public", "private", "protected", "static", "final", "const",
143
- "void", "int", "float", "double", "char", "bool", "string",
144
- "true", "false", "null", "None", "nil", "undefined",
145
- "if", "else", "elif", "then", "switch", "case", "default", "break",
146
- "for", "while", "do", "each", "in", "of", "to", "by",
147
- "try", "catch", "finally", "throw", "raise", "except",
148
- "return", "yield", "await", "async", "defer",
149
- "and", "or", "not", "is", "as", "with", "without",
150
- "lambda", "map", "filter", "reduce", "sort",
151
- "new", "delete", "free", "alloc", "realloc",
152
- "print", "printf", "println", "log", "debug", "error",
153
- "len", "size", "length", "count", "sum", "max", "min",
154
- "abs", "pow", "sqrt", "floor", "ceil", "round",
155
- "sin", "cos", "tan", "atan", "log", "exp",
156
- "zero", "one", "two", "three", "four", "five",
157
- "six", "seven", "eight", "nine", "ten",
158
- "first", "second", "third", "last", "next", "previous",
159
- "current", "initial", "final", "primary", "secondary",
160
- "main", "primary", "secondary", "basic", "advanced",
161
- "simple", "complex", "single", "double", "multiple",
162
- "add", "sub", "mul", "div", "mod", "inc", "dec",
163
- "push", "pop", "shift", "unshift", "insert", "remove",
164
- "append", "prepend", "concat", "join", "split", "slice",
165
- "open", "close", "read", "write", "load", "save",
166
- "create", "update", "delete", "insert", "select", "merge",
167
- "begin", "end", "start", "stop", "pause", "resume",
168
- "enable", "disable", "allow", "deny", "grant", "revoke",
169
- "user", "users", "name", "names", "id", "ids",
170
- "key", "keys", "value", "values", "field", "fields",
171
- "table", "tables", "row", "rows", "column", "columns",
172
- "index", "indexes", "indices", "query", "queries",
173
- "count", "avg", "total", "sum", "min", "max",
174
- "page", "pages", "home", "login", "logout", "signup",
175
- "error", "errors", "warning", "warnings", "info",
176
- "success", "failure", "status", "message", "messages",
177
- "request", "requests", "response", "responses",
178
- "client", "server", "api", "endpoint", "route",
179
- "http", "https", "url", "uri", "port", "host",
180
- "config", "configuration", "setting", "settings",
181
- "option", "options", "param", "params", "parameter", "parameters",
182
- "arg", "args", "argument", "arguments", "kwargs",
183
- "path", "dir", "directory", "dirs", "folder", "folders",
184
- "item", "items", "element", "elements", "entry", "entries",
185
- "note", "notes", "text", "texts", "content", "contents",
186
- "source", "sources", "target", "targets", "ref", "refs",
187
- "struct", "structs", "union", "unions", "enum", "enums",
188
- "impl", "implement", "implementation", "interface",
189
- "abstract", "virtual", "override", "overload",
190
- "base", "derived", "parent", "child", "children",
191
- "root", "leaf", "node", "nodes", "edge", "edges",
192
- "tree", "graph", "list", "queue", "stack", "heap",
193
- "map", "dict", "dictionary", "hash", "hashmap",
194
- "link", "links", "linked", "pointer", "pointers",
195
- "thread", "threads", "process", "processes", "task", "tasks",
196
- "sync", "async", "lock", "mutex", "semaphore",
197
- "buffer", "buffers", "cache", "cached", "pool", "pools",
198
- "memory", "disk", "network", "socket", "sockets",
199
- "stream", "streams", "packet", "packets", "frame",
200
- "meta", "metadata", "header", "headers", "body", "payload",
201
- "token", "tokens", "session", "cookie", "cookies",
202
- "auth", "login", "logout", "register", "password",
203
- "hash", "salt", "encrypt", "decrypt", "encode", "decode",
204
- "cert", "certificate", "key", "public", "private",
205
- "train", "training", "trained", "test", "testing", "tested",
206
- "valid", "validate", "validation", "eval", "evaluate",
207
- "model", "models", "layer", "layers", "weight", "weights",
208
- "bias", "biases", "loss", "losses", "grad", "grads",
209
- "lr", "learning_rate", "optimizer", "adam", "sgd",
210
- "batch", "batches", "epoch", "epochs", "step", "steps",
211
- "dataset", "dataloader", "tensor", "tensors",
212
- "gpu", "cpu", "tpu", "device", "devices", "memory",
213
- "math", "physics", "chemistry", "biology", "science",
214
- "compute", "calculate", "computation", "calculation",
215
- "equation", "formula", "expression", "theorem",
216
- "proof", "prove", "lemma", "axiom", "corollary",
217
- "function", "graph", "derivative", "integral", "limit",
218
- "sequence", "series", "matrix", "vector", "tensor",
219
- "set", "subset", "union", "intersection", "complement",
220
- "space", "group", "ring", "field", "module", "algebra",
221
- "analysis", "topology", "geometry", "statistics",
222
- "probability", "distribution", "random", "sample",
223
- "mean", "median", "mode", "variance", "std", "deviation",
224
- "linear", "nonlinear", "convex", "concave", "smooth",
225
- "algorithm", "algorithm", "complexity", "runtime",
226
- "tree", "graph", "sort", "search", "traverse",
227
- "recursive", "iterative", "dynamic", "greedy",
228
- "optimization", "constraint", "feasible", "optimal",
229
- "problem", "solution", "input", "output", "example",
230
- "question", "answer", "hint", "step", "reason",
231
- "think", "analyze", "approach", "solve", "verify",
232
- "check", "conclude", "summarize", "explain", "describe",
233
- "correct", "incorrect", "right", "wrong", "positive", "negative",
234
- "yes", "no", "maybe", "always", "never", "sometimes",
235
- ":", ";", ".", ",", "!", "?", "'", "\"", "(", ")", "[", "]", "{", "}",
236
- "<", ">", "=", "+", "-", "*", "/", "%", "&", "|", "^", "~",
237
- "@", "#", "$", "_", "`", "\\",
238
- "==", "!=", "<=", ">=", "&&", "||", "++", "--",
239
- "+=", "-=", "*=", "/=", "->", "=>", "::", "..",
240
- "...", "/*", "*/", "//", "<!--", "-->",
241
- "0", "1", "2", "3", "4", "5", "6", "7", "8", "9",
242
- "10", "11", "12", "13", "14", "15", "16", "17", "18", "19",
243
- "20", "30", "40", "50", "60", "70", "80", "90", "100",
244
- "-1", "-2", "0x", "0b", "0o",
245
- # GPT-2 style Ġ-prefixed versions for words preceded by space
246
- # The ByteLevel pre_tokenizer adds Ġ (U+0120) prefix after space
247
- "Ġthe", "Ġto", "Ġof", "Ġand", "Ġa", "Ġin", "Ġthat", "Ġis", "Ġwas", "Ġfor",
248
- "Ġon", "Ġwith", "Ġas", "Ġby", "Ġat", "Ġfrom", "Ġor", "Ġan", "Ġwill", "Ġwould",
249
- "Ġnot", "Ġbut", "Ġare", "Ġwere", "Ġbeen", "Ġhave", "Ġhas", "Ġhad", "Ġdo", "Ġdoes",
250
- "Ġdid", "Ġcan", "Ġcould", "Ġshould", "Ġmay", "Ġmight", "Ġshall", "Ġmust",
251
- "Ġif", "Ġelse", "Ġwhen", "Ġwhile", "Ġbecause", "Ġso", "Ġthen", "Ġthan",
252
- "Ġalso", "Ġeven", "Ġonly", "Ġjust", "Ġvery", "Ġtoo", "Ġstill", "Ġalready",
253
- "Ġhere", "Ġthere", "Ġwhere", "Ġwhen", "Ġwhy", "Ġhow", "Ġwhat", "Ġwhich",
254
- "Ġthis", "Ġthat", "Ġthese", "Ġthose", "Ġit", "Ġits", "Ġthey", "Ġthem",
255
- "Ġwe", "Ġus", "Ġour", "Ġyou", "Ġyour", "Ġhe", "Ġhim", "Ġhis",
256
- "Ġshe", "Ġher", "Ġhers", "Ġone", "Ġno", "Ġall", "Ġany", "Ġsome",
257
- "Ġeach", "Ġevery", "Ġboth", "Ġneither", "Ġeither", "Ġmore", "Ġmost", "Ġfew",
258
- "Ġother", "Ġanother", "Ġsuch", "Ġsame", "Ġdifferent", "Ġown",
259
- "Ġlike", "Ġwell", "Ġgood", "Ġbad", "Ġbetter", "Ġbest",
260
- "Ġnew", "Ġold", "Ġbig", "Ġsmall", "Ġlong", "Ġshort",
261
- "Ġhigh", "Ġlow", "Ġlarge", "Ġlittle", "Ġgreat", "Ġimportant",
262
- "Ġup", "Ġdown", "Ġin", "Ġout", "Ġon", "Ġoff", "Ġover", "Ġunder",
263
- "Ġagain", "Ġback", "Ġabout", "Ġaround", "Ġbetween", "Ġthrough",
264
- "Ġbefore", "Ġafter", "Ġduring", "Ġuntil", "Ġsince",
265
- "Ġfirst", "Ġlast", "Ġnext", "Ġprevious", "Ġfinal",
266
- "Ġget", "Ġgot", "Ġmake", "Ġmade", "Ġtake", "Ġtook", "Ġgive", "Ġgave",
267
- "Ġuse", "Ġused", "Ġusing", "Ġneed", "Ġneeds", "Ġneeded",
268
- "Ġwant", "Ġwants", "Ġwanted", "Ġlet", "Ġlets", "Ġlets",
269
- "Ġwork", "Ġworks", "Ġworked", "Ġworking", "Ġhelp", "Ġhelps", "Ġhelped",
270
- "Ġcall", "Ġcalls", "Ġcalled", "Ġcalling", "Ġset", "Ġsets", "Ġsetting",
271
- "Ġput", "Ġputs", "Ġputting", "Ġrun", "Ġruns", "Ġran", "Ġrunning",
272
- "Ġkeep", "Ġkeeps", "Ġkept", "Ġfind", "Ġfinds", "Ġfound", "Ġshow", "Ġshows",
273
- "Ġtry", "Ġtries", "Ġtried", "Ġtrying", "Ġstart", "��starts", "Ġstarted",
274
- "Ġchange", "Ġchanges", "Ġchanged", "Ġchanging", "Ġfollow", "Ġfollows",
275
- "Ġknow", "Ġknows", "Ġknown", "Ġthink", "Ġthinks", "Ġthought",
276
- "Ġsay", "Ġsays", "Ġsaid", "Ġsee", "Ġsees", "Ġsaw", "Ġseen",
277
- "Ġcome", "Ġcomes", "Ġcame", "Ġgo", "Ġgoes", "Ġwent", "Ġgone",
278
- "Ġbring", "Ġbrings", "Ġbrought", "Ġtell", "Ġtells", "Ġtold",
279
- "Ġlet", "Ġlets", "Ġleave", "Ġleaves", "Ġleft", "Ġhappen", "Ġhappens",
280
- "Ġprovide", "Ġprovides", "Ġprovided", "Ġconsider", "Ġconsiders",
281
- "Ġappear", "Ġappears", "Ġappeared", "Ġform", "Ġforms", "Ġformed",
282
- "Ġseem", "Ġseems", "Ġseemed", "Ġpoint", "Ġpoints", "Ġpointed",
283
- "Ġturn", "Ġturns", "Ġturned", "Ġturning", "Ġplay", "Ġplays",
284
- "Ġthings", "Ġthing", "Ġtime", "Ġtimes", "Ġyear", "Ġyears",
285
- "Ġpeople", "Ġplace", "Ġplaces", "Ġpart", "Ġparts", "Ġworld",
286
- "Ġnumber", "Ġnumbers", "Ġsystem", "Ġsystems", "Ġgroup", "Ġgroups",
287
- "Ġline", "Ġlines", "Ġfile", "Ġfiles", "Ġcode", "Ġdata",
288
- "Ġfunction", "Ġfunctions", "Ġvalue", "Ġvalues", "Ġtype", "Ġtypes",
289
- "Ġclass", "Ġclasses", "Ġobject", "Ġobjects", "Ġmethod", "Ġmethods",
290
- "Ġresult", "Ġresults", "Ġprocess", "Ġprocesses", "Ġstate", "Ġstates",
291
- "Ġmodel", "Ġmodels", "Ġlevel", "Ġlevels", "Ġcase", "Ġcases",
292
- "Ġexample", "Ġexamples", "Ġinput", "Ġinputs", "Ġoutput", "Ġoutputs",
293
- "Ġerror", "Ġerrors", "Ġstatus", "Ġmessage", "Ġmessages",
294
- "Ġrequest", "Ġrequests", "Ġresponse", "Ġresponses",
295
- "Ġreturn", "Ġreturns", "Ġimport", "Ġimports", "Ġexport", "Ġexports",
296
- "Ġdefine", "Ġdefines", "Ġdefined", "Ġdeclare", "Ġdeclares",
297
- "Ġinclude", "Ġincludes", "Ġconfig", "Ġsetting", "Ġsettings",
298
- "Ġoption", "Ġoptions", "Ġparam", "Ġparams", "Ġargs",
299
- "Ġtrue", "Ġfalse", "Ġnull", "ĠNone", "Ġundefined",
300
- "Ġlen", "Ġsize", "Ġcount", "Ġsum", "Ġmax", "Ġmin",
301
- "Ġprint", "Ġlog", "Ġdebug", "Ġinfo", "Ġwarn",
302
- "Ġtrain", "Ġtest", "Ġeval", "Ġvalid", "Ġval",
303
- "Ġpage", "Ġhome", "Ġname", "Ġkey", "Ġkeys",
304
- "Ġpath", "Ġdir", "Ġroot", "Ġitem", "Ġitems",
305
- "Ġsource", "Ġtarget", "Ġbase", "Ġmain", "Ġprimary",
306
- "Ġadd", "Ġremove", "Ġcreate", "Ġdelete", "Ġupdate",
307
- "Ġopen", "Ġclose", "Ġread", "Ġwrite", "Ġload", "Ġsave",
308
- "Ġpush", "Ġpop", "Ġinsert", "Ġappend", "Ġsplit",
309
- "Ġbegin", "Ġend", "Ġstart", "Ġstop", "Ġenable", "Ġdisable",
310
- "Ġuser", "Ġusers", "Ġadmin", "Ġmanager",
311
- "Ġstring", "Ġint", "Ġfloat", "Ġdouble", "Ġbool", "Ġvoid",
312
- "Ġlist", "Ġdict", "Ġset", "Ġtuple", "Ġarray",
313
- "Ġapi", "Ġurl", "Ġuri", "Ġendpoint", "Ġroute",
314
- "Ġhttp", "Ġhttps", "Ġclient", "Ġserver", "Ġsocket",
315
- "Ġmath", "Ġscience", "Ġdata", "Ġanalysis", "Ġtheory",
316
- "Ġproof", "Ġtheorem", "Ġlemma", "Ġequation", "Ġformula",
317
- "ing", "ed", "ly", "tion", "sion", "ment", "ness", "ity",
318
- "able", "ible", "al", "ial", "ical", "ous", "eous", "ious",
319
- "ive", "ative", "ful", "less", "like", "wise", "ward",
320
- "un", "re", "in", "im", "ir", "il", "dis", "mis", "non",
321
- "pre", "pro", "per", "trans", "inter", "intra", "extra",
322
- "sub", "super", "sur", "semi", "multi", "mono", "bi", "tri",
323
- "anti", "counter", "over", "under", "out", "up", "down",
324
- "co", "con", "com", "col", "cor", "de", "di", "dif",
325
- "ex", "extra", "fore", "macro", "micro", "mid", "mis",
326
- "out", "over", "post", "pre", "pro", "re", "semi", "sub",
327
- "super", "tele", "trans", "ultra", "un", "under", "up",
328
- "-ing", "-ed", "-ly", "-tion", "-sion", "-ment", "-ness", "-ity",
329
- "-able", "-ible", "-al", "-ous", "-ive", "-ful", "-less",
330
- "un-", "re-", "pre-", "non-", "anti-", "counter-",
331
- "self-", "all-", "well-", "so-", "to-", "in-",
332
- "'t", "'s", "'m", "'re", "'ve", "'ll", "'d",
333
- "n't", "don't", "can't", "won't", "isn't", "aren't",
334
- "wasn't", "weren't", "hasn't", "haven't", "hadn't",
335
- "doesn't", "didn't", "couldn't", "shouldn't", "wouldn't",
336
- "mustn't", "needn't", "mightn't",
337
- # Multi-char punctuation/symbols as single tokens
338
- "->", "=>", "<-", "<=", ">=", "==", "!=",
339
- "::", "..", "...", "/*", "*/", "//", "#",
340
- "\n", "\t",
341
- " ", " ", " ", " ",
342
- "Ċ", # byte-level newline
343
- "ĠĠ", "ĠĠĠ", "ĠĠĠĠ", # multiple spaces
344
- # Additional English words for 4096 vocab
345
- "about", "above", "across", "action", "actually", "address", "agree", "allow", "almost",
346
- "along", "already", "though", "although", "always", "American", "among", "amount",
347
- "animal", "another", "answer", "anything", "appear", "approach", "area", "argue",
348
- "arm", "article", "artist", "ask", "author", "available", "avoid", "away", "ball",
349
- "bank", "bar", "base", "battle", "beauty", "become", "became", "becoming", "bed",
350
- "behavior", "behind", "believe", "benefit", "best", "beyond", "bit", "black",
351
- "blood", "board", "body", "book", "born", "boss", "bother", "bottle", "bottom",
352
- "box", "boy", "brain", "break", "bridge", "brief", "bright", "bring", "broad",
353
- "brother", "budget", "build", "building", "burn", "business", "buy", "campaign",
354
- "capital", "car", "care", "career", "carry", "catch", "category", "cause",
355
- "central", "century", "certain", "chair", "chairman", "challenge", "chance",
356
- "character", "charge", "check", "choice", "choose", "chosen", "church", "citizen",
357
- "city", "civil", "claim", "clear", "clearly", "close", "club", "coach", "cold",
358
- "collection", "college", "color", "come", "comfortable", "comment", "committee",
359
- "common", "community", "company", "compare", "competition", "complete", "completely",
360
- "condition", "conference", "Congress", "connect", "conscious", "consider", "contain",
361
- "content", "continue", "contract", "control", "conversation", "cost", "could",
362
- "country", "couple", "course", "court", "cover", "create", "crime", "cultural",
363
- "culture", "cup", "current", "customer", "cut", "dark", "daughter", "deal",
364
- "death", "debate", "decade", "decide", "decision", "deep", "defense", "degree",
365
- "democrat", "democratic", "describe", "design", "despite", "detail", "determine",
366
- "develop", "development", "device", "die", "difference", "difficult", "dinner",
367
- "direction", "director", "discover", "discuss", "discussion", "disease", "dog",
368
- "door", "doubt", "down", "draw", "dream", "drive", "driver", "drop", "drug",
369
- "D", "early", "east", "eat", "economic", "economy", "edge", "edition", "editor",
370
- "education", "effect", "effort", "eight", "either", "election", "else", "employee",
371
- "encourage", "enemy", "energy", "enjoy", "enough", "enter", "entire", "environment",
372
- "environmental", "especially", "establish", "evening", "event", "ever", "everybody",
373
- "everyone", "everything", "evidence", "exactly", "examine", "example", "executive",
374
- "exist", "expect", "experience", "explain", "explanation", "extremely", "eye",
375
- "face", "fact", "factor", "fail", "fall", "family", "far", "fast", "father",
376
- "fear", "feature", "federal", "feel", "feeling", "field", "fight", "figure",
377
- "fill", "film", "final", "financial", "fine", "finish", "firm", "fish", "five",
378
- "floor", "fly", "focus", "follow", "food", "foot", "force", "foreign", "forget",
379
- "form", "former", "forward", "four", "free", "freedom", "friendly", "front",
380
- "full", "fund", "future", "game", "garden", "gas", "general", "generation",
381
- "gentleman", "girl", "glad", "glass", "goal", "god", "gold", "government",
382
- "governor", "great", "green", "ground", "group", "grow", "growth", "guess",
383
- "gun", "guy", "half", "hand", "handle", "hang", "happen", "happy", "hard",
384
- "head", "health", "hear", "heart", "heat", "heavy", "hell", "help", "here",
385
- "herself", "hide", "history", "hit", "hold", "home", "honest", "hope", "hospital",
386
- "hotel", "house", "huge", "human", "hundred", "husband", "idea", "identify",
387
- "image", "imagine", "impact", "implement", "imply", "important", "improve",
388
- "include", "including", "increase", "indeed", "indicate", "individual", "industry",
389
- "influence", "inform", "information", "inside", "instead", "institution",
390
- "interest", "international", "interview", "introduce", "investment", "involve",
391
- "issue", "item", "itself", "job", "join", "journal", "journey", "judge",
392
- "jump", "justice", "keep", "kill", "kind", "kitchen", "knowledge", "land",
393
- "language", "large", "last", "late", "later", "latter", "laugh", "launch",
394
- "law", "lawyer", "lay", "lead", "leader", "leading", "learn", "least", "leave",
395
- "left", "legal", "less", "let", "letter", "level", "lie", "life", "lift",
396
- "light", "likely", "limit", "line", "link", "list", "listen", "little", "live",
397
- "load", "local", "long", "look", "lord", "lose", "loss", "lost", "lot", "love",
398
- "low", "luck", "lunch", "machine", "main", "maintain", "major", "majority",
399
- "manage", "management", "manager", "manner", "manufacturer", "many", "map",
400
- "mark", "market", "marriage", "master", "material", "matter", "may", "maybe",
401
- "mean", "meaning", "measure", "media", "medical", "meet", "meeting", "member",
402
- "memory", "mention", "message", "method", "middle", "might", "military", "million",
403
- "mind", "minute", "miss", "mission", "mistake", "mix", "modern", "mom", "moment",
404
- "money", "month", "moral", "morning", "mother", "motion", "move", "movement",
405
- "movie", "music", "narrative", "nation", "national", "native", "natural", "nature",
406
- "near", "nearly", "necessarily", "necessary", "neck", "need", "negative", "neighbor",
407
- "neither", "network", "never", "nevertheless", "night", "none", "nor", "normal",
408
- "north", "note", "nothing", "notice", "notion", "now", "nowhere", "nuclear",
409
- "number", "occur", "ocean", "offer", "office", "officer", "official", "often",
410
- "oil", "OK", "old", "once", "online", "open", "operate", "operation", "opinion",
411
- "opportunity", "opposition", "option", "order", "organization", "original", "other",
412
- "otherwise", "ought", "outside", "overcome", "owner", "page", "pain", "paint",
413
- "pair", "paper", "parent", "park", "parliament", "part", "participant", "particular",
414
- "particularly", "partner", "party", "pass", "passage", "past", "path", "patient",
415
- "pattern", "pay", "peace", "pension", "people", "per", "percent", "perfect",
416
- "perform", "performance", "perhaps", "period", "permit", "person", "personal",
417
- "perspective", "phone", "physical", "pick", "picture", "piece", "place", "plan",
418
- "plant", "play", "player", "please", "pleasure", "plus", "pocket", "point",
419
- "police", "policy", "political", "politician", "politics", "pool", "poor",
420
- "popular", "population", "position", "positive", "possibility", "possible",
421
- "potentially", "power", "practice", "prepare", "presence", "present", "president",
422
- "pressure", "pretty", "prevent", "previous", "price", "primary", "principle",
423
- "prison", "private", "privilege", "probably", "problem", "procedure", "produce",
424
- "product", "production", "professional", "professor", "profile", "profit",
425
- "program", "project", "promise", "promote", "proper", "property", "proposal",
426
- "propose", "protect", "protection", "prove", "provide", "public", "publication",
427
- "publish", "pull", "purpose", "pursue", "push", "quality", "quarter", "question",
428
- "quick", "quickly", "quiet", "quite", "race", "radio", "raise", "range", "rate",
429
- "rather", "reach", "react", "reaction", "read", "reader", "reading", "ready",
430
- "real", "reality", "realize", "really", "reason", "reasonable", "receive",
431
- "recent", "recently", "recognize", "recommend", "record", "recover", "red",
432
- "reduce", "reflect", "reform", "region", "relate", "relationship", "relative",
433
- "relatively", "release", "relevant", "relief", "religion", "religious", "rely",
434
- "remain", "remember", "remind", "remove", "repeat", "replace", "report", "reporter",
435
- "represent", "representation", "republican", "reputation", "request", "require",
436
- "research", "resource", "respond", "response", "responsibility", "responsible",
437
- "rest", "restaurant", "result", "retain", "retire", "return", "reveal",
438
- "review", "revolution", "rich", "ride", "right", "ring", "rise", "risk", "river",
439
- "road", "rock", "role", "roll", "room", "rule", "run", "safe", "safety",
440
- "sale", "same", "sample", "save", "scale", "scene", "schedule", "school",
441
- "science", "scientist", "score", "screen", "sea", "search", "season", "seat",
442
- "second", "secret", "section", "security", "seed", "seek", "select", "self",
443
- "sell", "senate", "senator", "send", "sense", "serious", "serve", "service",
444
- "session", "settle", "seven", "sexual", "shadow", "shape", "share", "sharp",
445
- "sheet", "ship", "shock", "shoe", "shoot", "shop", "shot", "shoulder", "show",
446
- "shut", "sick", "side", "sight", "sign", "signal", "significance", "significant",
447
- "silence", "similar", "simple", "simply", "since", "sing", "single", "sister",
448
- "sit", "site", "situation", "six", "size", "skill", "skin", "small", "smile",
449
- "society", "soft", "soldier", "solid", "solution", "somebody", "somehow",
450
- "someone", "something", "sometimes", "somewhat", "son", "song", "soon", "sort",
451
- "sound", "source", "south", "space", "speak", "speaker", "special", "specific",
452
- "speech", "speed", "spend", "spin", "spirit", "spiritual", "split", "spokesman",
453
- "sport", "spot", "spread", "spring", "staff", "stage", "stand", "standard",
454
- "star", "start", "state", "statement", "station", "status", "stay", "step",
455
- "stick", "still", "stock", "stop", "store", "story", "straight", "strange",
456
- "strategic", "strategy", "street", "strength", "stress", "stretch", "strike",
457
- "strong", "structure", "struggle", "student", "study", "subject", "succeed",
458
- "success", "successful", "suddenly", "suffer", "sufficient", "suggest", "suggestion",
459
- "summer", "supply", "support", "suppose", "sure", "surface", "surgery", "surprise",
460
- "survey", "survive", "suspect", "sustain", "symbol", "system", "table", "talent",
461
- "talk", "tape", "target", "task", "taste", "tax", "teach", "teacher", "teaching",
462
- "team", "tear", "technical", "technique", "technology", "telephone", "television",
463
- "tell", "temperature", "tend", "term", "test", "testify", "testing", "text",
464
- "thank", "themselves", "therefore", "they", "thick", "thin", "thing", "think",
465
- "thinking", "third", "thirty", "threat", "threaten", "three", "throw", "thus",
466
- "ticket", "tight", "till", "time", "tiny", "tip", "title", "today", "together",
467
- "tomorrow", "tone", "tonight", "tool", "top", "total", "totally", "touch",
468
- "tough", "tour", "toward", "town", "track", "trade", "tradition", "traditional",
469
- "traffic", "train", "training", "transfer", "transform", "travel", "treat",
470
- "treatment", "tree", "trial", "trip", "troop", "trouble", "truck", "true",
471
- "truly", "trust", "truth", "try", "tube", "turn", "twice", "type", "typical",
472
- "uncle", "under", "understand", "understanding", "unfortunately", "union",
473
- "unique", "unit", "United", "universe", "university", "unless", "unlike",
474
- "unlikely", "unusual", "upper", "urban", "urge", "use", "used", "useful",
475
- "user", "usual", "usually", "value", "variety", "various", "vehicle", "version",
476
- "very", "veteran", "victim", "victory", "video", "view", "village", "violence",
477
- "visit", "voice", "volume", "vote", "voter", "wage", "wait", "walk", "wall",
478
- "want", "war", "warm", "warn", "warning", "wash", "watch", "water", "wave",
479
- "way", "weak", "weapon", "wear", "weather", "web", "wedding", "weekend",
480
- "weight", "welcome", "welfare", "well", "west", "western", "whatever", "wheel",
481
- "whenever", "whereas", "whether", "which", "while", "white", "whole", "whom",
482
- "whose", "wide", "widely", "wife", "wild", "will", "win", "wind", "window",
483
- "wine", "wing", "winner", "winter", "wire", "wish", "woman", "wonder", "wonderful",
484
- "wood", "word", "worker", "working", "works", "world", "worry", "worth", "would",
485
- "write", "writer", "writing", "wrong", "yard", "yeah", "year", "yet", "youth",
486
- "zone", "ability", "abroad", "absent", "absolute", "absorb", "abstract", "abuse",
487
- "academic", "accept", "access", "accident", "accompany", "accomplish", "account",
488
- "accurate", "accuse", "achieve", "acknowledge", "acquire", "adapt", "addition",
489
- "adjust", "administration", "admit", "adopt", "advance", "advantage", "adventure",
490
- "advertise", "advice", "advise", "advocate", "affair", "affect", "afford",
491
- "agency", "agenda", "agent", "aggression", "aggressive", "aid", "aim", "air",
492
- "airport", "alarm", "alcohol", "alert", "alive", "alliance", "allocate", "ally",
493
- "alone", "alter", "alternative", "amaze", "ambition", "amendment", "amid",
494
- "amongst", "analysis", "analyst", "angle", "angry", "anniversary", "announce",
495
- "annual", "anticipate", "anxiety", "anxious", "apart", "apartment", "apologize",
496
- "apparent", "appeal", "appearance", "appetite", "apple", "applicant", "application",
497
- "apply", "appoint", "appreciate", "appropriate", "approval", "approve", "architecture",
498
- "archive", "argue", "argument", "arrange", "arrangement", "arrest", "arrival",
499
- "arrive", "arrow", "articulate", "artificial", "aside", "aspect", "assault",
500
- "assemble", "assembly", "assert", "assess", "assessment", "asset", "assign",
501
- "assist", "assistance", "associate", "association", "assume", "assumption",
502
- "atmosphere", "attach", "attack", "attempt", "attend", "attention", "attitude",
503
- "attorney", "attract", "attraction", "attractive", "attribute", "audience",
504
- "auto", "automatic", "automatically", "autonomy", "available", "avenue", "average",
505
- "award", "aware", "awareness", "awful", "background", "bacteria", "balance",
506
- "bare", "barely", "barrier", "basic", "basis", "basket", "bath", "battery",
507
- "battle", "bay", "beach", "bean", "bear", "beat", "beautiful", "bedroom",
508
- "beer", "beginning", "behalf", "behave", "behavior", "being", "belief", "believable",
509
- "bell", "belong", "bench", "bend", "beneath", "beneficial", "beside", "bet",
510
- "betray", "bible", "bicycle", "bid", "bike", "bill", "bind", "biological",
511
- "biology", "birth", "biscuit", "bishop", "bite", "bitter", "blade", "blame",
512
- "blank", "blast", "bleed", "blend", "bless", "blind", "block", "blow", "blue",
513
- "blur", "board", "boast", "boat", "bomb", "bond", "bone", "bonus", "boom",
514
- "boost", "border", "bore", "borrow", "bottom", "bound", "boundary", "bowl",
515
- "brain", "branch", "brand", "brave", "bread", "breadth", "breast", "breath",
516
- "breathe", "breathing", "breed", "brick", "bride", "bridge", "briefly", "brilliant",
517
- "broadcast", "broken", "bronze", "brow", "brown", "brush", "bubble", "bucket",
518
- "buddy", "buffalo", "bunch", "burden", "burglar", "burn", "burst", "bury",
519
- "bus", "butter", "button", "cabin", "cabinet", "cable", "cake", "calculate",
520
- "calculation", "calendar", "calm", "camera", "camp", "campus", "canal", "cancel",
521
- "candidate", "candle", "cap", "capable", "capacity", "captain", "capture",
522
- "carbon", "card", "careful", "carefully", "carrier", "carry", "cart", "carve",
523
- "cast", "castle", "casualty", "catalog", "catalogue", "catch", "cattle",
524
- "celebrate", "celebration", "cell", "cellular", "census", "centimeter",
525
- "ceremony", "certainly", "certificate", "chain", "chair", "chairman",
526
- "chamber", "champion", "championship", "channel", "chapter", "characteristic",
527
- "charge", "charity", "chart", "chase", "cheap", "cheat", "cheek", "cheese",
528
- "chemical", "chemistry", "chest", "chicken", "chief", "childhood", "chip",
529
- "chocolate", "chorus", "christian", "Christmas", "chronic", "chunk", "circle",
530
- "circuit", "circumstance", "cite", "citizen", "civilian", "claim",
531
- "clarify", "clarity", "clash", "classic", "classical", "classification",
532
- "classroom", "clause", "clean", "clear", "clever", "click", "client",
533
- "cliff", "climate", "climb", "clinic", "clinical", "clock", "clone",
534
- "closed", "closely", "closer", "closet", "closing", "cloth", "clothe",
535
- "clothes", "clothing", "cloud", "club", "cluster", "coal", "coalition",
536
- "coast", "coat", "code", "coffee", "cognitive", "coin", "cold",
537
- "collapse", "collar", "colleague", "collect", "collection", "collective",
538
- "colonial", "colony", "color", "column", "combat", "combine", "combined",
539
- "comedy", "comfort", "command", "commander", "comment", "commerce",
540
- "commercial", "commission", "commit", "commitment", "commodity", "communicate",
541
- "communication", "communist", "compact", "companion", "comparison", "compel",
542
- "compensate", "compensation", "compete", "competition", "competitive",
543
- "competitor", "complain", "complaint", "complement", "complex", "complexity",
544
- "complicate", "complicated", "comply", "component", "compose", "composition",
545
- "compound", "comprehensive", "comprise", "compromise", "compulsory", "compute",
546
- "computer", "conceal", "concede", "conceive", "concentrate", "concentration",
547
- "concept", "conception", "concern", "concerning", "concert", "conclude",
548
- "conclusion", "concrete", "condemn", "conduct", "conference", "confess",
549
- "confession", "confidence", "confident", "confidential", "confine", "confirm",
550
- "conflict", "confront", "confusion", "congratulate", "congress", "connect",
551
- "connection", "conscious", "consciousness", "consecutive", "consensus",
552
- "consent", "consequence", "consequently", "conservation", "conservative",
553
- "considerable", "considerably", "consist", "consistent", "consistently",
554
- "constant", "constantly", "constitute", "constitution", "constitutional",
555
- "construct", "construction", "consult", "consultant", "consume", "consumer",
556
- "consumption", "contact", "contemporary", "contend", "contest", "context",
557
- "continent", "continually", "continuity", "continuous", "continuously",
558
- "contradiction", "contrary", "contribute", "contribution", "contributor",
559
- "controversial", "controversy", "convenience", "convenient", "convention",
560
- "conventional", "conversation", "conversion", "convert", "convey", "convict",
561
- "conviction", "convince", "cook", "cookie", "cool", "cooperate", "cooperation",
562
- "coordinate", "coordination", "cope", "copper", "copy", "copyright", "core",
563
- "corn", "corner", "corporate", "corporation", "correct", "correction",
564
- "correctly", "correlate", "correlation", "correspond", "correspondent",
565
- "corridor", "corruption", "costly", "cotton", "council", "counsel",
566
- "counselor", "count", "counter", "counterpart", "county", "coup",
567
- "courage", "cousin", "cover", "coverage", "crack", "craft", "crash",
568
- "creative", "creativity", "creator", "creature", "credibility", "credit",
569
- "creep", "crew", "crime", "criminal", "crisis", "criterion", "critic",
570
- "critical", "criticism", "criticize", "crop", "cross", "crowd", "crown",
571
- "crucial", "crude", "cruel", "cruise", "crush", "cry", "crystal", "cube",
572
- "cuisine", "cultivate", "curious", "currency", "current", "curriculum",
573
- "curtain", "curve", "custody", "custom", "customary", "customer", "cutting",
574
- "cycle", "dad", "damage", "damn", "dance", "danger", "dangerous",
575
- "dare", "data", "database", "dawn", "dead", "deadline", "deadly",
576
- "deaf", "deal", "dealer", "dear", "debate", "debt", "decade",
577
- "decay", "deceive", "decent", "decide", "decision", "decisive", "deck",
578
- "declaration", "declare", "decline", "decorate", "decrease", "decree",
579
- "dedicate", "deem", "defeat", "defend", "defendant", "defender", "defense",
580
- "defensive", "deficit", "define", "definite", "definitely", "definition",
581
- "defy", "degree", "delay", "delegate", "delegation", "delete", "deliberate",
582
- "deliberately", "delicate", "delicious", "delight", "deliver", "delivery",
583
- "demand", "democracy", "democrat", "democratic", "demographic", "demonstrate",
584
- "demonstration", "denial", "denote", "deny", "depart", "department", "departure",
585
- "depend", "dependence", "dependent", "depict", "deposit", "depress", "depression",
586
- "deprive", "depth", "deputy", "derive", "descend", "describe", "description",
587
- "desert", "deserve", "design", "designate", "designer", "desirable",
588
- "desire", "desk", "desperate", "desperately", "despite", "destination",
589
- "destroy", "destruction", "detail", "detailed", "detain", "detect",
590
- "detection", "detective", "detention", "deteriorate", "determination",
591
- "determine", "determined", "develop", "development", "developmental",
592
- "devote", "devote", "diabetes", "diagnose", "diagnosis", "dialogue",
593
- "diameter", "diamond", "diary", "dictate", "diet", "differ", "difference",
594
- "differentiate", "differently", "difficulty", "dig", "digest", "digital",
595
- "dignity", "dilemma", "dimension", "diminish", "dinner", "dioxide",
596
- "dip", "diplomat", "diplomatic", "direct", "direction", "directly",
597
- "director", "dirty", "disability", "disable", "disadvantage",
598
- "disagree", "disappear", "disappoint", "disappointment", "disaster",
599
- "disastrous", "disc", "discharge", "discipline", "disclose", "discount",
600
- "discourse", "discover", "discovery", "discrepancy", "discretion",
601
- "discrimination", "discuss", "discussion", "disease", "dismiss", "disorder",
602
- "dispatch", "display", "disposal", "dispose", "dispute", "disrupt",
603
- "dissolve", "distance", "distant", "distinct", "distinction", "distinctive",
604
- "distinguish", "distort", "distract", "distress", "distribute", "distribution",
605
- "distributor", "district", "disturb", "dive", "diverse", "diversity",
606
- "divide", "division", "divorce", "dock", "doctor", "doctrine", "document",
607
- "documentary", "dollar", "domain", "dome", "domestic", "dominant",
608
- "dominate", "donation", "donor", "dose", "dot", "double", "doubt",
609
- "doubtful", "downtown", "draft", "drag", "drain", "drama", "dramatic",
610
- "dramatically", "drastic", "draw", "drawing", "drink", "drive",
611
- "driver", "drop", "drought", "drown", "drum", "drunk", "dry",
612
- "dual", "dubious", "duck", "due", "dull", "dump", "durable", "duration",
613
- "dust", "duty", "dynamic", "dynamics", "eager", "eagle", "ear",
614
- "earning", "earth", "ease", "easily", "eastern", "echo", "eclipse",
615
- "ecological", "ecology", "economics", "economist", "economy",
616
- "ecosystem", "edit", "edition", "editor", "editorial", "educate",
617
- "education", "educational", "educator", "effective", "effectively",
618
- "effectiveness", "efficiency", "efficient", "efficiently", "effort",
619
- "elaborate", "elbow", "elderly", "elect", "election", "electoral",
620
- "electric", "electrical", "electricity", "electronic", "electronics",
621
- "elegant", "element", "elementary", "eliminate", "elimination", "elite",
622
- "elsewhere", "email", "embargo", "embark", "embarrass", "embassy",
623
- "embed", "embody", "embrace", "emerge", "emergence", "emergency",
624
- "emission", "emotion", "emotional", "emphasis", "emphasize", "empire",
625
- "empirical", "employ", "employee", "employer", "employment", "empower",
626
- "enable", "enact", "encompass", "encounter", "encourage", "encouragement",
627
- "endanger", "endeavor", "endorse", "endorsement", "endure", "enforce",
628
- "enforcement", "engage", "engagement", "engine", "engineering", "enhance",
629
- "enjoy", "enjoyment", "enlarge", "enormous", "enrich", "enroll",
630
- "ensemble", "ensure", "enter", "enterprise", "entertain", "entertainment",
631
- "enthusiasm", "enthusiast", "enthusiastic", "entirely", "entitle",
632
- "entity", "entrepreneur", "entry", "envelope", "environment",
633
- "environmental", "epidemic", "episode", "equal", "equality", "equation",
634
- "equip", "equipment", "equivalent", "era", "erect", "error", "erupt",
635
- "escalate", "escape", "especially", "essay", "essence", "essential",
636
- "essentially", "establish", "establishment", "estate", "estimate",
637
- "estimation", "eternal", "ethical", "ethics", "ethnic", "evacuate",
638
- "evaluate", "evaluation", "evenly", "event", "eventually", "ever",
639
- "everyday", "evidence", "evident", "evil", "evoke", "evolution",
640
- "evolutionary", "evolve", "exact", "exaggerate", "examination",
641
- "examine", "example", "exceed", "excellence", "excellent", "exception",
642
- "exceptional", "excess", "excessive", "exchange", "excite", "excitement",
643
- "exciting", "exclude", "exclusion", "exclusive", "exclusively",
644
- "excuse", "execute", "execution", "executive", "exemplify",
645
- "exercise", "exert", "exhaust", "exhibit", "exhibition", "exile",
646
- "exist", "existence", "exit", "expand", "expansion", "expect",
647
- "expectation", "expedition", "expel", "expenditure", "expense",
648
- "expensive", "expert", "expertise", "explain", "explanation",
649
- "explicit", "explicitly", "explode", "exploit", "exploitation",
650
- "exploration", "explore", "explosion", "explosive", "export",
651
- "expose", "exposure", "express", "expression", "extend", "extension",
652
- "extensive", "extensively", "extent", "external", "extinct",
653
- "extinction", "extra", "extract", "extraordinary", "extreme",
654
- "extremely", "eye", "fabric", "fabulous", "facade", "face",
655
- "facial", "facilitate", "facility", "fact", "faction", "faculty",
656
- "fade", "fail", "failure", "fair", "fairly", "fairness", "faith",
657
- "faithful", "fake", "fame", "familiar", "famine", "fan", "fancy",
658
- "fantasy", "fare", "fascinate", "fascinating", "fashion", "fat",
659
- "fate", "fatigue", "fault", "favor", "favorable", "favorite",
660
- "fax", "fear", "feasible", "feast", "feather", "federal",
661
- "federation", "fee", "feed", "feedback", "feel", "feeling",
662
- "fellow", "fellowship", "female", "fence", "fertile", "fertilizer",
663
- "festival", "fetch", "fever", "fiber", "fiction", "field", "fierce",
664
- "fifteen", "fifty", "fig", "fight", "fighter", "figure", "file",
665
- "fill", "filter", "final", "finally", "finance", "financial",
666
- "financially", "financing", "finding", "finger", "finished",
667
- "fire", "firm", "firmly", "fiscal", "fish", "fisherman", "fishing",
668
- "fitness", "fix", "fixture", "flag", "flame", "flash", "flat",
669
- "flavor", "flee", "fleet", "flesh", "flexibility", "flexible",
670
- "flight", "float", "flock", "flood", "floor", "flour", "flow",
671
- "flower", "fluid", "flush", "fly", "focus", "folk", "football",
672
- "footnote", "footstep", "forbid", "forbidden", "forecast", "forehead",
673
- "foreign", "foreigner", "forest", "forever", "forge", "forget",
674
- "forgive", "fork", "form", "formal", "format", "formation", "former",
675
- "formula", "formulate", "fort", "forth", "fortunate", "fortune",
676
- "forum", "forward", "fossil", "foster", "found",
677
- "foundation", "founder", "fountain", "fraction", "fracture", "fragile",
678
- "fragment", "frame", "framework", "franchise", "frank", "frankly",
679
- "fraud", "free", "freedom", "freely", "freeze", "freight",
680
- "frequency", "frequent", "frequently", "fresh", "freshman", "friction",
681
- "friendly", "friendship", "frighten", "frog", "front",
682
- "frontier", "frost", "frown", "frozen", "fruit", "frustrate",
683
- "frustration", "fuel", "fulfill", "full", "fun", "function",
684
- "functional", "fund", "fundamental", "funding", "funeral", "funny",
685
- "fur", "furious", "furniture", "further", "furthermore", "fury",
686
- "fusion", "future", "gain", "galaxy", "gallery", "gallon", "gambling",
687
- "gap", "garage", "garbage", "garden", "garlic", "garment", "gas",
688
- "gasoline", "gate", "gather", "gathering", "gauge", "gaze", "gear",
689
- "gender", "gene", "general", "generally", "generate", "generation",
690
- "generator", "generous", "genetic", "genetics", "genius", "genocide",
691
- "genre", "gentle", "gentleman", "gently", "genuine", "genuinely",
692
- "gesture", "giant", "gift", "gigantic", "glimpse", "global",
693
- "globalization", "globe", "glory", "glove", "glow", "glucose",
694
- "goal", "goddess", "gold", "golden", "golf", "goodness",
695
- "goods", "gorgeous", "gospel", "gossip", "govern", "governance",
696
- "government", "governor", "grab", "grace", "grade", "gradually",
697
- "graduate", "graduation", "grain", "gram", "grammar", "grand",
698
- "grandfather", "grandmother", "grant", "graph", "graphic", "grasp",
699
- "grass", "grateful", "grave", "gravity", "gray", "greatly",
700
- "green", "greenhouse", "greet", "grief", "grin", "grind", "grip",
701
- "grocery", "gross", "ground", "groundwater", "growth", "guarantee",
702
- "guard", "guardian", "guess", "guest", "guidance", "guide",
703
- "guideline", "guilty", "guitar", "gulf", "gun", "gut", "guy",
704
- "gym", "habit", "habitat", "hair", "half", "hall", "halt", "hammer",
705
- "hand", "handful", "handle", "handling", "handwriting", "handy",
706
- "hang", "happen", "happiness", "harassment", "harbor", "hardly",
707
- "hardware", "harm", "harmful", "harmony", "harvest", "hat",
708
- "hate", "haul", "hay", "hazard", "head", "headache", "headline",
709
- "headquarters", "heal", "health", "healthcare", "healthy",
710
- "heap", "hearing", "heart", "heat", "heating", "heaven",
711
- "heavily", "heavy", "hedge", "heel", "height", "helicopter",
712
- "hell", "helmet", "helpful", "herb", "heritage", "hero",
713
- "heroin", "herself", "hesitate", "hidden", "hide", "hierarchy",
714
- "highlight", "highly", "highway", "hike", "hill", "himself",
715
- "hip", "hire", "historian", "historic", "historical", "history",
716
- "hit", "hobby", "hold", "holder", "holding", "hole",
717
- "holiday", "hollow", "holy", "homeland", "homeless", "homework",
718
- "honest", "honesty", "honey", "honor", "hook", "hope",
719
- "hopeful", "hopefully", "horizon", "horizontal", "hormone",
720
- "horn", "horrible", "horror", "horse", "hospitality", "host",
721
- "hostage", "hostile", "hot", "hotline", "hour", "housing",
722
- "hover", "human", "humane", "humanitarian", "humanity",
723
- "humor", "hundred", "hunger", "hungry", "hunt", "hunter",
724
- "hunting", "hurt", "husband", "hut", "hydrogen", "hygiene",
725
- "hypothesis", "ice", "icon", "idea", "ideal", "identical",
726
- "identification", "identify", "identity", "ideology", "ignorance",
727
- "ignore", "ill", "illegal", "illness", "illusion", "illustrate",
728
- "illustration", "image", "imaginary", "imagination", "imagine",
729
- "imitate", "immediate", "immediately", "immense", "immigrant",
730
- "immigration", "immune", "immunity", "impact", "implement",
731
- "implementation", "implication", "implicit", "imply", "import",
732
- "importance", "impose", "impossible", "impress", "impression",
733
- "impressive", "imprison", "improbable", "improve", "improvement",
734
- "impulse", "inability", "inappropriate", "incentive", "incidence",
735
- "incident", "inclination", "incline", "include", "including",
736
- "inclusion", "inclusive", "income", "incorporate", "incorrect",
737
- "increase", "increasingly", "incredible", "incur", "indeed",
738
- "independence", "independent", "independently", "index",
739
- "indicate", "indication", "indicator", "indictment",
740
- "indigenous", "indirect", "indispensable", "individual",
741
- "individuality", "indoor", "induce", "indulge", "industrial",
742
- "industrialize", "industry", "inequality", "inevitable",
743
- "inevitably", "infant", "infect", "infection", "infer",
744
- "inference", "inferior", "infinite", "infinity", "inflation",
745
- "inflict", "influence", "influential", "info", "inform",
746
- "informal", "information", "infrastructure", "ingredient",
747
- "inhabit", "inhabitant", "inherent", "inherit", "inhibit",
748
- "initial", "initially", "initiate", "initiative", "inject",
749
- "injection", "injure", "injury", "inmate", "inner",
750
- "innocent", "innovation", "innovative", "input", "inquiry",
751
- "insect", "insert", "insertion", "inside", "insight",
752
- "insist", "inspect", "inspection", "inspector", "inspiration",
753
- "inspire", "install", "installation", "installment", "instance",
754
- "instant", "instantly", "instead", "instinct",
755
- "institute", "institution", "institutional", "instruct",
756
- "instruction", "instructor", "instrument", "instrumental",
757
- "insufficient", "insult", "insurance", "intact", "integral",
758
- "integrate", "integration", "integrity", "intellectual",
759
- "intelligence", "intelligent", "intend", "intense",
760
- "intensity", "intensive", "intent", "intention", "intentional",
761
- "interact", "interaction", "interactive", "interest",
762
- "interested", "interesting", "interface", "interfere",
763
- "interference", "interim", "interior", "intermediate",
764
- "internal", "international", "internet", "interpret",
765
- "interpretation", "interrupt", "interruption",
766
- "intersection", "interval", "intervene", "intervention",
767
- "interview", "intimate", "intrigue", "intrinsic", "introduce",
768
- "introduction", "intuition", "intuitive", "invade",
769
- "invasion", "invent", "invention", "inventory", "invest",
770
- "investigate", "investigation", "investigator", "investment",
771
- "investor", "invisible", "invitation", "invite", "involve",
772
- "involvement", "iron", "irony", "irrelevant", "irrigation",
773
- "island", "isolate", "isolation", "issue", "item",
774
- "itself", "ivory", "jail", "jam", "jet", "jewel",
775
- "jewelry", "job", "join", "joint", "jointly", "joke",
776
- "journal", "journalism", "journalist", "journey", "joy",
777
- "judge", "judgment", "judicial", "juice", "jump", "junction",
778
- "jungle", "junior", "jurisdiction", "jury", "justice",
779
- "justification", "justify", "juvenile", "keen", "keeper",
780
- "kettle", "key", "keyboard", "kick", "kid", "kidnap",
781
- "kidney", "kin", "kindness", "king", "kingdom", "kiss",
782
- "kit", "kitchen", "knee", "kneel", "knife", "knock",
783
- "knot", "label", "labor", "laboratory", "lace", "lack",
784
- "ladder", "lady", "lake", "lamp", "land", "landing",
785
- "landlord", "landmark", "landscape", "lane", "lap", "largely",
786
- "laser", "late", "latte", "latter", "laugh", "laughter",
787
- "launch", "laundry", "lavatory", "law", "lawn", "lawmaker",
788
- "lawn", "lawsuit", "lawyer", "layout", "leading", "leaf",
789
- "league", "leak", "lean", "leap", "learner", "learning",
790
- "lease", "leather", "leave", "lecture",
791
- "legacy", "legend", "legendary", "legislation", "legislative",
792
- "legislature", "legitimate", "leisure", "lemon", "lend",
793
- "length", "lens", "lesson", "lest", "lethal", "letter",
794
- "lettuce", "liberal", "liberation", "liberty", "library",
795
- "license", "lid", "lie", "lifelong", "lifestyle", "lifetime",
796
- "lift", "light", "lighting", "lightly", "likelihood",
797
- "likewise", "limb", "limit", "limitation", "limited",
798
- "limitless", "limp", "line", "linear", "linen", "liner",
799
- "linger", "linguistic", "lining", "link", "lion",
800
- "lip", "liquid", "liquor", "list", "listen", "listener",
801
- "listing", "liter", "literally", "literary", "literature",
802
- "litigation", "liver", "living", "load", "loan",
803
- "lobby", "local", "locate", "location", "lock", "lodge",
804
- "log", "logic", "logical", "logo", "lonely", "longitudinal",
805
- "lookout", "loop", "loose", "loosen", "lord", "lose",
806
- "loss", "loud", "lounge", "lovely", "lover",
807
- "low", "lower", "loyal", "loyalty", "luck", "lucky",
808
- "luggage", "lump", "lunch", "lung", "luxury",
809
- "lyric", "machine", "machinery", "mad", "magazine", "magic",
810
- "magical", "magnetic", "magnificent", "magnitude", "maid",
811
- "mail", "mainland", "mainly", "mainstream", "maintain",
812
- "maintenance", "majesty", "majority", "maker",
813
- "makeup", "male", "mall", "mama", "mammal", "manage",
814
- "manageable", "management", "manager", "mandate",
815
- "mandatory", "maneuver", "manifest", "manipulate",
816
- "manipulation", "mankind", "manuscript", "maple",
817
- "marathon", "marble", "march", "margin", "marginal",
818
- "marine", "mark", "marker", "market", "marketing",
819
- "marketplace", "marriage", "married", "marry", "mask",
820
- "mass", "massacre", "massive", "master", "masterpiece",
821
- "match", "mate", "material", "maternal", "math",
822
- "mathematical", "mathematics", "matter", "mature",
823
- "maturity", "maximize", "maximum", "mayor", "meadow",
824
- "meaning", "meaningful", "means", "meantime",
825
- "measurable", "measure", "measurement", "mechanic",
826
- "mechanical", "mechanism", "medal", "media",
827
- "mediate", "mediation", "medicaid", "medical",
828
- "medication", "medicine", "medieval", "meditation",
829
- "medium", "meet", "melody", "melt", "member",
830
- "membership", "memo", "memoir", "memorandum",
831
- "memorial", "memorize", "menace",
832
- "mental", "mentally", "mention", "mentor", "menu",
833
- "merchandise", "merchant", "mercy", "mere", "merely",
834
- "merge", "merger", "merit", "merry", "mess",
835
- "messenger", "metal", "metaphor", "method",
836
- "methodology", "metric", "metropolitan",
837
- "microphone", "microscope", "midday", "middle",
838
- "midnight", "midst", "migrant", "migrate", "migration",
839
- "mild", "mile", "milestone", "militant", "military",
840
- "militia", "mill", "millennium", "millimeter",
841
- "mineral", "mingle", "miniature", "minimal",
842
- "minimize", "minimum", "mining", "minister",
843
- "ministry", "minor", "minority", "minute",
844
- "miracle", "mirror", "miserable", "misery",
845
- "misleading", "missile", "missing", "mission",
846
- "missionary", "mist", "mistake", "mistaken",
847
- "mistress", "misunderstand", "misunderstanding",
848
- "mixture", "moan", "mobile", "mobility",
849
- "mobilize", "mode", "moderate", "moderately",
850
- "moderation", "modern", "modest", "modification",
851
- "modify", "module", "moisture", "molecule",
852
- "molest", "moment", "momentum", "monarchy",
853
- "monastery", "monitor", "monk", "monopoly",
854
- "monster", "monument", "mood", "moon",
855
- "moral", "morale", "morality", "moreover",
856
- "mortal", "mortality", "mortgage", "mosaic",
857
- "mosque", "mostly", "mother", "motion",
858
- "motivate", "motivation", "motive", "motor",
859
- "motorcycle", "mount", "mountain", "mounting",
860
- "mourn", "mouse", "mouth", "movement",
861
- "movie", "mud", "multiple", "multiplication",
862
- "multiply", "multitude", "municipal",
863
- "municipality", "murder", "murderer", "murmur",
864
- "muscle", "muscular", "museum", "mushroom",
865
- "musical", "musician", "muslim", "mutual",
866
- "mutually", "mysterious", "mystery", "myth",
867
- "mythology", "nail", "naked", "narrative",
868
- "narrow", "nasty", "nation", "national",
869
- "nationalism", "nationalist", "nationality",
870
- "nationwide", "native", "natural", "naturally",
871
- "nature", "naval", "navigation", "navy",
872
- "nearby", "neat", "necessarily", "necessary",
873
- "necessity", "neck", "necklace", "needle",
874
- "negative", "neglect", "negotiate",
875
- "negotiation", "negotiator", "neighbor",
876
- "neighborhood", "neither", "nerve", "nervous",
877
- "nest", "net", "network", "neutral", "nevertheless",
878
- "niche", "nickel", "niece", "night", "nightmare",
879
- "nitrogen", "noble", "nobody", "nod", "noise",
880
- "noisy", "nominal", "nominate", "nomination",
881
- "nominee", "nonprofit", "nonsense", "norm",
882
- "normal", "normally", "normative", "north",
883
- "northeast", "northern", "northwest", "notable",
884
- "notably", "notation", "notebook", "nothing",
885
- "notice", "noticeable", "notification",
886
- "notion", "notorious", "novel", "novelist",
887
- "novelty", "nowhere", "nuclear", "nuance",
888
- "nucleus", "nuisance", "number", "numerical",
889
- "numerous", "nurse", "nursery", "nursing",
890
- "nutrient", "nutrition", "nutritional",
891
- "nutritious", "nylon", "oak", "obedience",
892
- "obedient", "obese", "obesity", "obey",
893
- "objection", "objective", "obligation",
894
- "oblige", "obscure", "observation", "observe",
895
- "observer", "obsession", "obstacle",
896
- "obtain", "obvious", "obviously", "occasion",
897
- "occasional", "occasionally", "occupation",
898
- "occupy", "occur", "occurrence",
899
- "offend", "offense", "offensive", "offer",
900
- "offering", "officer", "official", "officially",
901
- "offspring", "olive", "omission",
902
- "omit", "ongoing", "onion", "onset",
903
- "opening", "openly", "opera", "operate",
904
- "operating", "operation", "operational",
905
- "operator", "opinion", "opponent",
906
- "opportunity", "oppose", "opposite",
907
- "opposition", "opt", "optical", "optimism",
908
- "optimist", "optimistic", "optimum",
909
- "option", "optional", "oral", "orbit",
910
- "orchestra", "ordeal", "orderly",
911
- "ordinarily", "ordinary", "organ",
912
- "organic", "organization", "organizational",
913
- "organize", "organized", "organizer",
914
- "orientation", "origin", "original",
915
- "originally", "originate", "ornament",
916
- "orthodox", "other", "otherwise",
917
- "ought", "ounce", "outbreak",
918
- "outcome", "outdoor", "outer", "outfit",
919
- "outing", "outlet", "outline", "outlook",
920
- "output", "outrage", "outright",
921
- "outset", "outside", "outsider", "outstanding",
922
- "outward", "oval", "oven", "overall",
923
- "overcome", "overlook", "overnight",
924
- "override", "overseas", "oversee",
925
- "overturn", "overwhelm", "overwhelming",
926
- "owe", "own", "owner", "ownership",
927
- "oxygen", "ozone", "pace", "pack",
928
- "package", "packaging", "packet", "pad",
929
- "paddle", "page", "pain", "painful",
930
- "paint", "painter", "painting", "pair",
931
- "palace", "pale", "palm", "pan",
932
- "panel", "panic", "paper", "parade",
933
- "paradigm", "paradise", "paradox",
934
- "paragraph", "parallel", "parameter",
935
- "parcel", "pardon", "parent",
936
- "parental", "parish", "park",
937
- "parking", "parliament", "parliamentary",
938
- "partial", "partially", "participant",
939
- "participate", "participation",
940
- "particle", "particular", "particularly",
941
- "partly", "partner", "partnership",
942
- "passage", "passenger", "passing",
943
- "passion", "passionate", "passive",
944
- "passport", "password", "past",
945
- "paste", "pastor", "patch", "patent",
946
- "path", "pathway", "patience",
947
- "patient", "patrol", "patron",
948
- "pattern", "pause", "pave", "pavement",
949
- "paw", "payment", "peace", "peaceful",
950
- "peak", "peasant", "peculiar", "pedestrian",
951
- "peer", "penalty", "pencil", "penetrate",
952
- "peninsula", "pension", "people",
953
- "pepper", "perceive", "percentage",
954
- "perception", "perfect", "perfectly",
955
- "perform", "performance", "performer",
956
- "perfume", "perhaps", "period",
957
- "periodic", "peripheral", "permanent",
958
- "permanently", "permission", "permit",
959
- "persist", "persistence", "persistent",
960
- "persona", "personal", "personality",
961
- "personally", "personnel",
962
- "perspective", "persuade", "pet",
963
- "petition", "petroleum", "phase",
964
- "phenomenon", "philosopher", "philosophical",
965
- "philosophy", "phone",
966
- "photograph", "photographer", "photographic",
967
- "photography", "phrase", "physical",
968
- "physically", "physician", "physics",
969
- "piano", "pick", "picture",
970
- "piece", "pierce", "pig", "pile",
971
- "pillar", "pillow", "pilot",
972
- "pin", "pine", "pink", "pioneer",
973
- "pipe", "pit", "pitch", "pizza",
974
- "placement", "plain", "plaintiff",
975
- "plantation", "plate", "platform",
976
- "plausible", "playback", "playful",
977
- "playground", "plead", "pleasant",
978
- "pledge", "plenty", "plot",
979
- "plug", "plunge", "plural", "plus",
980
- "pocket", "poem", "poet", "poetic",
981
- "poetry", "poison", "poisonous",
982
- "polar", "pole", "police",
983
- "policeman", "policy", "polish",
984
- "polite", "political", "politically",
985
- "politician", "politics",
986
- "poll", "pollution", "pond",
987
- "pony", "pool", "pop",
988
- "pope", "popular", "popularity",
989
- "population", "porch", "port",
990
- "portable", "porter", "portion",
991
- "portrait", "portray", "pose",
992
- "position", "positive", "positively",
993
- "possess", "possession", "possessive",
994
- "possibility", "possible", "possibly",
995
- "postage", "postal", "poster",
996
- "potato", "potent", "potential",
997
- "potentially", "pottery",
998
- "poverty", "powder", "powerful",
999
- "practically", "practice",
1000
- "practitioner", "praise", "pray",
1001
- "prayer", "preach", "precede",
1002
- "precedent", "precious", "precise",
1003
- "precisely", "precision", "predict",
1004
- "predictable", "prediction",
1005
- "predominantly", "preface", "prefer",
1006
- "preference", "pregnancy", "pregnant",
1007
- "prejudice", "preliminary",
1008
- "premier", "premise", "premium",
1009
- "preparation", "prepare", "prepared",
1010
- "prescription", "presence",
1011
- "presentation", "presently",
1012
- "preservation", "preserve",
1013
- "presidency", "president", "presidential",
1014
- "pressing", "pressure",
1015
- "presumably", "presume", "pretend",
1016
- "pretty", "prevail", "prevalence",
1017
- "prevalent", "prevention",
1018
- "preview", "previous", "previously",
1019
- "prey", "pricing", "pride",
1020
- "priest", "primarily", "primary",
1021
- "prime", "primitive", "prince",
1022
- "princess", "principal", "principle",
1023
- "print", "printer", "printing",
1024
- "prior", "priority", "prison",
1025
- "prisoner", "privacy",
1026
- "privilege", "privileged",
1027
- "prize", "proactive", "probable",
1028
- "probably", "probe", "problem",
1029
- "problematic", "procedural",
1030
- "procedure", "proceed", "proceeding",
1031
- "proceeds", "processor",
1032
- "proclaim", "produce", "producer",
1033
- "productive", "productivity",
1034
- "profession", "professional",
1035
- "professor", "proficiency",
1036
- "profile", "profit",
1037
- "profitable", "profound",
1038
- "program", "programming",
1039
- "progressive", "prohibit",
1040
- "prohibition", "project",
1041
- "projection", "prominent",
1042
- "promise", "promising",
1043
- "promote", "promotion",
1044
- "prompt", "proof",
1045
- "propaganda", "proper", "properly",
1046
- "property", "prophet",
1047
- "proportion", "proposal",
1048
- "propose", "proposed",
1049
- "prosecution", "prosecutor",
1050
- "prospect", "prosperity",
1051
- "protect", "protection",
1052
- "protective", "protein",
1053
- "protest", "protester",
1054
- "protocol", "proud",
1055
- "prove", "proverb",
1056
- "provide", "provided",
1057
- "province", "provincial",
1058
- "provision", "provoke",
1059
- "proxy", "prudent",
1060
- "psychiatric", "psychiatry",
1061
- "psychological", "psychologist",
1062
- "psychology", "pub",
1063
- "publication", "publicity",
1064
- "publicly", "publish",
1065
- "publisher", "publishing",
1066
- "pulse", "pump",
1067
- "punch", "punish",
1068
- "punishment", "pupil",
1069
- "purchase", "purchaser",
1070
- "pure", "purely",
1071
- "purify", "purple",
1072
- "purpose", "purse",
1073
- "pursue", "pursuit",
1074
- "puzzle", "qualification",
1075
- "qualified", "qualify",
1076
- "qualitative", "quantify",
1077
- "quantitative", "quarterly",
1078
- "queen", "quest",
1079
- "questionable", "questionnaire",
1080
- "queue", "quit",
1081
- "quiz", "quota",
1082
- "quotation", "quote",
1083
- "rabbit", "racial",
1084
- "racism", "racist",
1085
- "radiation", "radical",
1086
- "rage", "raid",
1087
- "rail", "railroad",
1088
- "railway", "rain",
1089
- "rainbow", "rally",
1090
- "random", "range",
1091
- "rank", "ranking",
1092
- "rape", "rapid",
1093
- "rapidly", "rare",
1094
- "rarely", "rat",
1095
- "rate", "rating",
1096
- "ratio", "rational",
1097
- "raw", "ray",
1098
- "react", "reaction",
1099
- "readily", "reading",
1100
- "realistic", "reality",
1101
- "realization", "realize",
1102
- "realm", "rear",
1103
- "reason", "reasonable",
1104
- "reasonably", "reasoning",
1105
- "reassure", "rebel",
1106
- "rebellion", "rebuild",
1107
- "recall", "receipt",
1108
- "receiver", "recent",
1109
- "recently", "reception",
1110
- "recession", "recipe",
1111
- "recipient", "reckon",
1112
- "recognition", "recognize",
1113
- "recommend", "recommendation",
1114
- "reconcile", "reconstruction",
1115
- "recording", "recover",
1116
- "recovery", "recreation",
1117
- "recruit", "recruitment",
1118
- "reduction", "redundant",
1119
- "reef", "refer",
1120
- "referee", "reference",
1121
- "referendum", "referral",
1122
- "reflection", "reform",
1123
- "refrain", "refresh",
1124
- "refuge", "refugee",
1125
- "refund", "refusal",
1126
- "refuse", "regain",
1127
- "regard", "regarding",
1128
- "regardless", "regime",
1129
- "regiment", "regional",
1130
- "register", "registration",
1131
- "regret", "regular",
1132
- "regularly", "regulate",
1133
- "regulation", "regulator",
1134
- "regulatory", "rehabilitation",
1135
- "reign", "reinforce",
1136
- "reinforcement", "reject",
1137
- "rejection", "relate",
1138
- "relation", "relationship",
1139
- "relative", "relatively",
1140
- "relax", "relaxation",
1141
- "release", "relevant",
1142
- "reliability", "reliable",
1143
- "relief", "relieve",
1144
- "religion", "religious",
1145
- "reluctant", "reluctantly",
1146
- "remainder", "remains",
1147
- "remark", "remarkable",
1148
- "remarkably", "remedy",
1149
- "reminder", "remnant",
1150
- "remote", "removal",
1151
- "remove", "renaissance",
1152
- "render", "renew",
1153
- "renewable", "renewal",
1154
- "rent", "rental",
1155
- "repair", "repay",
1156
- "repeat", "repeatedly",
1157
- "repent", "repetition",
1158
- "replace", "replacement",
1159
- "reportedly", "reporter",
1160
- "representation", "representative",
1161
- "repression", "reprint",
1162
- "reproduce", "reproduction",
1163
- "republic", "republican",
1164
- "reputation", "request",
1165
- "require", "requirement",
1166
- "rescue", "resemble",
1167
- "resentment", "reservation",
1168
- "reserve", "reservoir",
1169
- "reside", "residence",
1170
- "residence", "resident",
1171
- "residential", "residual",
1172
- "resign", "resignation",
1173
- "resist", "resistance",
1174
- "resistant", "resolution",
1175
- "resolve", "resort",
1176
- "resource", "respect",
1177
- "respectable", "respectful",
1178
- "respective", "respectively",
1179
- "respond", "respondent",
1180
- "response", "responsibility",
1181
- "responsible", "responsive",
1182
- "restoration", "restore",
1183
- "restrain", "restraint",
1184
- "restrict", "restriction",
1185
- "restrictive", "restructuring",
1186
- "resume", "retail",
1187
- "retailer", "retain",
1188
- "retaliation", "retire",
1189
- "retirement", "retreat",
1190
- "retrieval", "retrieve",
1191
- "revelation", "revenge",
1192
- "revenue", "reverse",
1193
- "review", "revise",
1194
- "revision", "revival",
1195
- "revive", "revolution",
1196
- "revolutionary", "reward",
1197
- "rhetoric", "rhythm",
1198
- "rib", "ribbon",
1199
- "rid", "ride",
1200
- "ridge", "ridiculous",
1201
- "rifle", "rigid",
1202
- "riot", "rip",
1203
- "ripple", "ritual",
1204
- "rival", "rivalry",
1205
- "roar", "robbery",
1206
- "robe", "robot",
1207
- "robust", "rocket",
1208
- "rod", "roll",
1209
- "roller", "romance",
1210
- "romantic", "rookie",
1211
- "rope", "rose",
1212
- "rotate", "rotation",
1213
- "rotten", "rough",
1214
- "roughly", "round",
1215
- "route", "routine",
1216
- "row", "royal",
1217
- "royalty", "rub",
1218
- "rubber", "rug",
1219
- "ruin", "ruler",
1220
- "ruling", "rumor",
1221
- "runner", "running",
1222
- "rural", "rush",
1223
- "sacred", "sacrifice",
1224
- "sad", "saddle",
1225
- "sadly", "sadness",
1226
- "sake", "salad",
1227
- "salary", "sale",
1228
- "salmon", "salon",
1229
- "salt", "salute",
1230
- "salvation", "sample",
1231
- "sanction", "sanctuary",
1232
- "sand", "satellite",
1233
- "satisfaction", "satisfactory",
1234
- "satisfy", "sauce",
1235
- "saving", "savings",
1236
- "scale", "scandal",
1237
- "scare", "scared",
1238
- "scary", "scatter",
1239
- "scenario", "scent",
1240
- "schedule", "scheme",
1241
- "scholar", "scholarship",
1242
- "schooling", "scientific",
1243
- "scientist", "scope",
1244
- "score", "scorn",
1245
- "scrap", "scream",
1246
- "screen", "screening",
1247
- "screw", "script",
1248
- "scrutiny", "seal",
1249
- "search", "season",
1250
- "seasonal", "seating",
1251
- "secular", "secure",
1252
- "security", "seed",
1253
- "seek", "segment",
1254
- "seize", "seizure",
1255
- "select", "selection",
1256
- "selective", "self",
1257
- "seller", "senate",
1258
- "senator", "senior",
1259
- "sensation", "sensational",
1260
- "sensitivity", "sensor",
1261
- "sentence", "sentiment",
1262
- "separate", "separation",
1263
- "sequence", "sequencing",
1264
- "serial", "series",
1265
- "session", "settle",
1266
- "settlement", "settler",
1267
- "setup", "severe",
1268
- "severely", "severity",
1269
- "sew", "sewage",
1270
- "shade", "shadow",
1271
- "shaft", "shake",
1272
- "shall", "shallow",
1273
- "shame", "shape",
1274
- "shareholder", "sharing",
1275
- "shark", "shed",
1276
- "sheer", "sheet",
1277
- "shelf", "shell",
1278
- "shelter", "shield",
1279
- "shift", "shine",
1280
- "shipment", "shipping",
1281
- "shirt", "shock",
1282
- "shopping", "shortage",
1283
- "shortly", "shot",
1284
- "shoulder", "shout",
1285
- "shove", "shower",
1286
- "shrimp", "shrink",
1287
- "shrub", "shrug",
1288
- "shutter", "sibling",
1289
- "sickness", "sideways",
1290
- "siege", "sigh",
1291
- "sight", "sign",
1292
- "signal", "signature",
1293
- "significance", "significant",
1294
- "significantly", "silence",
1295
- "silent", "silicon",
1296
- "silk", "silly",
1297
- "silver", "similar",
1298
- "similarity", "similarly",
1299
- "simmer", "simplicity",
1300
- "simplify", "simply",
1301
- "simulation", "simultaneously",
1302
- "sin", "sincere",
1303
- "sincerely", "singer",
1304
- "single", "singular",
1305
- "sink", "sip",
1306
- "sister", "situation",
1307
- "sizable", "size",
1308
- "sketch", "ski",
1309
- "skilled", "skillful",
1310
- "skim", "skin",
1311
- "skip", "skirt",
1312
- "skull", "slap",
1313
- "slash", "slave",
1314
- "slavery", "sleep",
1315
- "sleeve", "slice",
1316
- "slide", "slight",
1317
- "slightly", "slim",
1318
- "slip", "slogan",
1319
- "slope", "slot",
1320
- "slow", "slowly",
1321
- "smart", "smell",
1322
- "smile", "smoke",
1323
- "smooth", "smoothly",
1324
- "snap", "sneak",
1325
- "snapshot", "snow",
1326
- "soak", "soap",
1327
- "soar", "soccer",
1328
- "social", "socialism",
1329
- "socialist", "societal",
1330
- "society", "sociological",
1331
- "sociology", "soda",
1332
- "software", "soil",
1333
- "solar", "soldier",
1334
- "sole", "solely",
1335
- "solemn", "solicitor",
1336
- "solidarity", "solitary",
1337
- "solo", "soluble",
1338
- "solution", "solve",
1339
- "somebody", "somehow",
1340
- "someone", "sometime",
1341
- "somewhat", "song",
1342
- "sophisticated", "sore",
1343
- "sorrow", "sort",
1344
- "soul", "sound",
1345
- "soup", "sour",
1346
- "source", "southeast",
1347
- "southern", "southwest",
1348
- "sovereign", "sovereignty",
1349
- "sow", "spacecraft",
1350
- "spacing", "span",
1351
- "spare", "spark",
1352
- "speak", "speaker",
1353
- "spear", "special",
1354
- "specialist", "specialize",
1355
- "specialty", "species",
1356
- "specific", "specifically",
1357
- "specification", "specify",
1358
- "specimen", "spectacle",
1359
- "spectacular", "spectator",
1360
- "spectrum", "speculate",
1361
- "speculation", "speech",
1362
- "spell", "spelling",
1363
- "spend", "sphere",
1364
- "spill", "spin",
1365
- "spine", "spiral",
1366
- "spirit", "spiritual",
1367
- "spite", "splash",
1368
- "split", "spokesman",
1369
- "spokesperson", "spokeswoman",
1370
- "sponsor", "sponsorship",
1371
- "spontaneous", "spoon",
1372
- "sport", "spot",
1373
- "spouse", "spread",
1374
- "spring", "sprint",
1375
- "spur", "spy",
1376
- "squad", "squadron",
1377
- "square", "squeeze",
1378
- "stability", "stabilize",
1379
- "stable", "stadium",
1380
- "staff", "stage",
1381
- "stake", "stakeholder",
1382
- "stall", "stance",
1383
- "stand", "standard",
1384
- "standing", "staple",
1385
- "stare", "stark",
1386
- "startup", "starvation",
1387
- "starve", "statement",
1388
- "statue", "status",
1389
- "statute", "statutory",
1390
- "steady", "steal",
1391
- "steam", "steel",
1392
- "steep", "steer",
1393
- "stem", "stereotype",
1394
- "sterling", "stern",
1395
- "steward", "stick",
1396
- "sticky", "stiff",
1397
- "stimulate", "stimulus",
1398
- "stir", "stitch",
1399
- "stock", "stomach",
1400
- "stoppage", "storage",
1401
- "storm", "story",
1402
- "strain", "strand",
1403
- "strap", "strategic",
1404
- "strategically", "strategist",
1405
- "strategy", "straw",
1406
- "stream", "street",
1407
- "strength", "strengthen",
1408
- "stress", "stretch",
1409
- "strict", "strictly",
1410
- "stride", "strike",
1411
- "striker", "string",
1412
- "strip", "stripe",
1413
- "strive", "stroke",
1414
- "stronghold", "strongly",
1415
- "structural", "structure",
1416
- "struggle", "stubborn",
1417
- "studio", "study",
1418
- "stuff", "stumble",
1419
- "stun", "stunning",
1420
- "stupid", "style",
1421
- "subject", "subjective",
1422
- "sublime", "submission",
1423
- "submit", "subordinate",
1424
- "subsequent", "subsequently",
1425
- "subsidy", "substance",
1426
- "substantial", "substantially",
1427
- "substantive", "substitute",
1428
- "substitution", "subtle",
1429
- "subtlety", "subtract",
1430
- "suburb", "suburban",
1431
- "subversion", "subvert",
1432
- "succeed", "success",
1433
- "successful", "successfully",
1434
- "succession", "successive",
1435
- "successor", "suck",
1436
- "sudden", "suddenly",
1437
- "sue", "suffer",
1438
- "suffering", "sufficient",
1439
- "sufficiently", "sugar",
1440
- "suicide", "suit",
1441
- "suitable", "suite",
1442
- "sulfur", "sum",
1443
- "summarize", "summary",
1444
- "summit", "sunlight",
1445
- "sunny", "sunrise",
1446
- "sunset", "sunshine",
1447
- "superb", "superficial",
1448
- "superintendent", "superior",
1449
- "superiority", "supermarket",
1450
- "supervise", "supervision",
1451
- "supervisor", "supplement",
1452
- "supplementary", "supplier",
1453
- "supply", "support",
1454
- "supporter", "supportive",
1455
- "suppose", "supposedly",
1456
- "suppress", "suppression",
1457
- "supreme", "surcharge",
1458
- "surface", "surge",
1459
- "surgeon", "surgery",
1460
- "surgical", "surname",
1461
- "surpass", "surplus",
1462
- "surprise", "surprised",
1463
- "surprising", "surprisingly",
1464
- "surrender", "surround",
1465
- "surrounding", "surveillance",
1466
- "survey", "survival",
1467
- "survive", "survivor",
1468
- "susceptible", "suspect",
1469
- "suspend", "suspense",
1470
- "suspension", "suspicion",
1471
- "suspicious", "sustain",
1472
- "sustainable", "sustained",
1473
- "swap", "swear",
1474
- "sweep", "sweet",
1475
- "swell", "swift",
1476
- "swim", "swimming",
1477
- "swing", "switch",
1478
- "sword", "symbol",
1479
- "symbolic", "symmetry",
1480
- "sympathetic", "sympathy",
1481
- "symphony", "symptom",
1482
- "syndrome", "synthesis",
1483
- "synthetic", "system",
1484
- "systematic", "tackle",
1485
- "tactical", "tactics",
1486
- "tag", "tail",
1487
- "takeover", "tale",
1488
- "talent", "talented",
1489
- "tank", "tap",
1490
- "tape", "target",
1491
- "tariff", "task",
1492
- "taste", "tax",
1493
- "taxation", "taxpayer",
1494
- "teaching", "tear",
1495
- "tease", "technical",
1496
- "technically", "technician",
1497
- "technique", "technological",
1498
- "technology", "teenage",
1499
- "teenager", "telecommunications",
1500
- "telegraph", "telephone",
1501
- "telescope", "television",
1502
- "temper", "temperature",
1503
- "temple", "temporarily",
1504
- "temporary", "tempt",
1505
- "temptation", "tenant",
1506
- "tendency", "tender",
1507
- "tennis", "tension",
1508
- "tent", "tenure",
1509
- "terminal", "terminate",
1510
- "termination", "term",
1511
- "terrain", "terrible",
1512
- "terribly", "terrific",
1513
- "territorial", "territory",
1514
- "terror", "terrorism",
1515
- "terrorist", "testament",
1516
- "testify", "testimony",
1517
- "textbook", "textile",
1518
- "texture", "thankful",
1519
- "theater", "theatre",
1520
- "theft", "theological",
1521
- "theology", "theoretical",
1522
- "theorist", "theory",
1523
- "therapist", "therapy",
1524
- "thereafter", "thereby",
1525
- "thermal", "thesis",
1526
- "thickness", "thief",
1527
- "thigh", "thin",
1528
- "thinking", "thirst",
1529
- "thirsty", "thorn",
1530
- "thorough", "thoroughly",
1531
- "thoughtful", "thoughtless",
1532
- "thriller", "thrive",
1533
- "throat", "throne",
1534
- "thrust", "thumb",
1535
- "thunder", "tide",
1536
- "timber", "timely",
1537
- "timing", "tissue",
1538
- "title", "toe",
1539
- "tolerance", "tolerant",
1540
- "tolerate", "toll",
1541
- "tomato", "ton",
1542
- "tone", "tongue",
1543
- "tool", "tooth",
1544
- "topic", "topical",
1545
- "torch", "torture",
1546
- "total", "totally",
1547
- "touch", "tourism",
1548
- "tourist", "tournament",
1549
- "tow", "towel",
1550
- "tower", "toxic",
1551
- "trace", "track",
1552
- "tractor", "trade",
1553
- "trademark", "trader",
1554
- "trading", "tradition",
1555
- "traditional", "traditionally",
1556
- "traffic", "tragedy",
1557
- "tragic", "trail",
1558
- "trainer", "training",
1559
- "trait", "transaction",
1560
- "transcript", "transfer",
1561
- "transform", "transformation",
1562
- "transit", "transition",
1563
- "translate", "translation",
1564
- "translator", "transmission",
1565
- "transmit", "transparency",
1566
- "transparent", "transplant",
1567
- "transport", "transportation",
1568
- "trap", "trash",
1569
- "trauma", "traumatic",
1570
- "traveler", "treasure",
1571
- "treat", "treatment",
1572
- "treaty", "tremendous",
1573
- "trend", "trial",
1574
- "triangle", "tribal",
1575
- "tribe", "tribunal",
1576
- "trigger", "trim",
1577
- "triumph", "troop",
1578
- "trophy", "tropical",
1579
- "trouble", "troublesome",
1580
- "trunk", "trust",
1581
- "trustee", "truth",
1582
- "tube", "tuition",
1583
- "tumor", "tune",
1584
- "tunnel", "turmoil",
1585
- "turnover", "turtle",
1586
- "tutor", "tutorial",
1587
- "twist", "typical",
1588
- "typically", "tyranny",
1589
- "ugly", "ultimate",
1590
- "ultimately", "ultimatum",
1591
- "umbrella", "unable",
1592
- "unacceptable", "unanimous",
1593
- "unaware", "uncertainty",
1594
- "uncomfortable", "unconscious",
1595
- "unconstitutional", "undergo",
1596
- "undergraduate", "underground",
1597
- "underline", "underlying",
1598
- "undermine", "underneath",
1599
- "underscore", "undertake",
1600
- "undertaking", "underwear",
1601
- "undoubtedly", "unemployment",
1602
- "unexpected", "unexpectedly",
1603
- "unfair", "unfold",
1604
- "unfortunate", "unfortunately",
1605
- "unhappy", "unhealthy",
1606
- "unified", "uniform",
1607
- "unilateral", "unintended",
1608
- "union", "unique",
1609
- "unity", "universal",
1610
- "universally", "universe",
1611
- "unknown", "unlawful",
1612
- "unlike", "unlikely",
1613
- "unnecessary", "unpleasant",
1614
- "unprecedented", "unrest",
1615
- "unveil", "upcoming",
1616
- "update", "upgrade",
1617
- "uphold", "upset",
1618
- "upturn", "uranium",
1619
- "urban", "urge",
1620
- "urgency", "urgent",
1621
- "usage", "utilize",
1622
- "utmost", "utter",
1623
- "utterly", "vacancy",
1624
- "vacation", "vaccination",
1625
- "vaccine", "vacuum",
1626
- "valid", "validate",
1627
- "validity", "valley",
1628
- "valuable", "valuation",
1629
- "valve", "variable",
1630
- "variation", "varied",
1631
- "vary", "vast",
1632
- "vector", "vegetable",
1633
- "vegetation", "vehicle",
1634
- "veil", "vein",
1635
- "velocity", "vendor",
1636
- "venture", "venue",
1637
- "verbal", "verdict",
1638
- "verify", "versatile",
1639
- "verse", "version",
1640
- "versus", "vertical",
1641
- "vessel", "veteran",
1642
- "veterinary", "viable",
1643
- "vibrant", "vibration",
1644
- "vice", "victim",
1645
- "victorious", "victory",
1646
- "video", "view",
1647
- "viewer", "viewpoint",
1648
- "vigorous", "village",
1649
- "violate", "violation",
1650
- "violence", "violent",
1651
- "virgin", "virtual",
1652
- "virtually", "virtue",
1653
- "virus", "visa",
1654
- "visible", "vision",
1655
- "visual", "vital",
1656
- "vitamin", "vivid",
1657
- "vocabulary", "vocal",
1658
- "vocational", "voice",
1659
- "volatile", "volcano",
1660
- "volume", "voluntary",
1661
- "volunteer", "vote",
1662
- "voter", "voting",
1663
- "vulnerability", "vulnerable",
1664
- "wage", "wagon",
1665
- "waist", "wallet",
1666
- "wander", "ward",
1667
- "warehouse", "warfare",
1668
- "warmth", "warn",
1669
- "warning", "warrant",
1670
- "warranty", "warrior",
1671
- "wary", "waterfall",
1672
- "waterproof", "watershed",
1673
- "wave", "wavelength",
1674
- "wax", "weakness",
1675
- "wealth", "wealthy",
1676
- "weapon", "wear",
1677
- "weary", "weather",
1678
- "weave", "web",
1679
- "website", "wedding",
1680
- "weed", "weekday",
1681
- "weekend", "weekly",
1682
- "weigh", "weight",
1683
- "weird", "welcome",
1684
- "welfare", "wellness",
1685
- "whatsoever", "wheel",
1686
- "whenever", "whereas",
1687
- "whereby", "wherein",
1688
- "whichever", "whisper",
1689
- "white", "whoever",
1690
- "wholesale", "wholly",
1691
- "widespread", "widow",
1692
- "width", "willing",
1693
- "willingness", "wisdom",
1694
- "withdraw", "withdrawal",
1695
- "wither", "withhold",
1696
- "within", "without",
1697
- "witness", "wolf",
1698
- "wonder", "wooden",
1699
- "wool", "workforce",
1700
- "workout", "workplace",
1701
- "workshop", "worship",
1702
- "worst", "worthwhile",
1703
- "worthy", "wound",
1704
- "wrap", "wrist",
1705
- "writer", "writing",
1706
- "wrongly", "yacht",
1707
- "yell", "young",
1708
- "youngster", "zone",
1709
  ]
1710
 
1711
- # Exclude byte-level chars (they already have IDs 51-306)
1712
- byte_chars = {byte2char[b] for b in range(256)}
1713
-
1714
- # Deduplicate while preserving order
1715
- seen = set()
1716
- deduped = []
1717
- for w in common_words:
1718
- if w not in seen and w not in byte_chars:
1719
- seen.add(w)
1720
- deduped.append(w)
1721
-
1722
- # Truncate to fit total vocab <= 4096 (IDs 307-4095 = 3789 slots)
1723
- max_words = 4096 - 307
1724
- if len(deduped) > max_words:
1725
- deduped = deduped[:max_words]
1726
- common_words = deduped
1727
-
1728
- next_id = 307
1729
- word_tokens = []
1730
- for w in common_words:
1731
- word_tokens.append((w, next_id))
1732
- next_id += 1
1733
-
1734
- # Build vocab dict
1735
- vocab = {}
1736
- for name, idx in special_tokens:
1737
- if name in vocab:
1738
- print(f"DUPLICATE: {name!r} @ {vocab[name]} and {idx}")
1739
- vocab[name] = idx
1740
- for char, idx in byte_tokens:
1741
- if char in vocab:
1742
- print(f"DUPLICATE BYTE: {char!r} (U+{ord(char):04X}) @ {vocab[char]} and {idx}")
1743
- vocab[char] = idx
1744
- for name, idx in word_tokens:
1745
- if name in vocab:
1746
- print(f"DUPLICATE WORD: {name!r} @ {vocab[name]} and {idx}")
1747
- vocab[name] = idx
1748
-
1749
- # Build added_tokens
1750
- added_tokens = []
1751
- for name, idx in special_tokens:
1752
- added_tokens.append({
1753
- "id": idx,
1754
- "content": name,
1755
- "single_word": False,
1756
- "lstrip": False,
1757
- "rstrip": False,
1758
- "normalized": False,
1759
- "special": True,
1760
- })
1761
-
1762
- # Build tokenizer.json
1763
- tokenizer_json = {
1764
- "version": "1.0",
1765
- "truncation": None,
1766
- "padding": None,
1767
- "added_tokens": added_tokens,
1768
- "normalizer": None,
1769
- "pre_tokenizer": {
1770
- "type": "ByteLevel",
1771
- "add_prefix_space": False,
1772
- "trim_offsets": True,
1773
- },
1774
- "post_processor": None,
1775
- "decoder": {
1776
- "type": "ByteLevel",
1777
- "add_prefix_space": False,
1778
- "trim_offsets": True,
1779
- },
1780
- "model": {
1781
- "type": "BPE",
1782
- "dropout": None,
1783
- "unk_token": "<unk>",
1784
- "byte_fallback": True,
1785
- "vocab": vocab,
1786
- "merges": [],
1787
- },
1788
- }
1789
 
1790
  if __name__ == "__main__":
1791
- with open("tokenizer/tokenizer.json", "w", encoding="utf-8") as f:
1792
- json.dump(tokenizer_json, f, ensure_ascii=False, separators=(",", ":"))
1793
-
1794
- with open("tokenizer/special_tokens_map.json", "w", encoding="utf-8") as f:
1795
- json.dump({
1796
- "bos_token": "<s>",
1797
- "eos_token": "</s>",
1798
- "pad_token": "<pad>",
1799
- "unk_token": "<unk>",
1800
- }, f, ensure_ascii=False, indent=2)
1801
-
1802
- import os
1803
- size = os.path.getsize("tokenizer/tokenizer.json")
1804
- print(f"tokenizer.json: {size:,} bytes")
1805
- print(f"Vocab entries: {len(vocab)}")
1806
- print(f"Added tokens: {len(added_tokens)}")
1807
- print(f"Merges: 0")
1808
- print(f"Byte fallback: True")
1809
- assert len(vocab) <= 4096, f"Vocab size overflow: {len(vocab)}"
1810
- print(f"Free slots: {4096 - len(vocab)}")
1811
-
1812
- from transformers import PreTrainedTokenizerFast
1813
- from tokenizers import Tokenizer as Tk
1814
- tok_obj = Tk.from_file("tokenizer/tokenizer.json")
1815
- tok = PreTrainedTokenizerFast(tokenizer_object=tok_obj)
1816
- tok.add_special_tokens({"pad_token": "<pad>", "bos_token": "<s>", "eos_token": "</s>", "unk_token": "<unk>"})
1817
- test = "Hello, how are you?"
1818
- enc = tok.encode(test)
1819
- dec = tok.decode(enc)
1820
- print(f"OK: {enc} -> {dec!r}")
1821
-
1822
- test2 = "def f(): return 42"
1823
- enc2 = tok.encode(test2)
1824
- dec2 = tok.decode(enc2)
1825
- print(f"OK: {enc2} -> {dec2!r}")
1826
-
1827
- enc3 = tok.encode("<|system|>Hi<|user|>there")
1828
- print(f"Special: {enc3} -> {tok.decode(enc3)!r}")
1829
-
1830
- test4 = "café résumé"
1831
- enc4 = tok.encode(test4)
1832
- dec4 = tok.decode(enc4)
1833
- print(f"Unicode: {enc4} -> {dec4!r}")
 
1
+ """Generate BPE tokenizer trained on textbook corpus.
2
+
3
+ Vocab layout: 0-50 special tokens, 51-306 byte-level chars, 307-4095 BPE merges.
4
+ """
5
  import json
6
+ import os
7
+ from tokenizers import Tokenizer, models, pre_tokenizers, trainers, decoders
8
 
9
+ TOKENIZER_DIR = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "tokenizer")
10
+ BOOKS_DIR = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "data", "books")
 
 
 
 
 
 
 
 
 
11
 
12
+ os.makedirs(TOKENIZER_DIR, exist_ok=True)
 
 
13
 
 
 
 
 
14
  special_tokens = [
15
+ "<unk>", "<s>", "</s>", "<pad>",
16
+ "<|system|>", "<|user|>", "<|assistant|>",
17
+ "<think>", "</think>",
18
+ "[INST]", "[/INST]",
19
+ "<|begin_of_thought|>", "<|end_of_thought|>",
20
+ "<|reflect|>", "<|revise|>", "<|verify|>",
21
+ "<|code|>", "<|text|>", "<|math|>",
22
+ "<|think|>", "<|answer|>", "<|step|>", "<|reason|>",
23
+ "<|check|>", "<|output|>", "<|plan|>", "<|solve|>",
24
+ "<|analyze|>", "<|conclude|>", "<|approach|>", "<|alternative|>",
25
+ "<|summary|>", "<|question|>", "<|hint|>", "<|example|>",
26
+ "<|correct|>", "<|incorrect|>", "<|feedback|>",
27
+ "<|start|>", "<|end|>", "<|sep|>", "<|cls|>",
28
+ "<|tool|>", "<|function|>", "<|result|>", "<|input|>",
29
+ "<|detect|>", "<|context|>", "<|proof|>", "<|lemma|>", "<|theorem|>",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
30
  ]
31
 
32
+ special_map = {t: i for i, t in enumerate(special_tokens)}
33
+ NUM_SPECIAL = len(special_tokens)
34
+
35
+ def find_book_files():
36
+ if not os.path.isdir(BOOKS_DIR):
37
+ print(f"Warning: {BOOKS_DIR} not found, using empty corpus")
38
+ return []
39
+ files = []
40
+ for fname in os.listdir(BOOKS_DIR):
41
+ if fname.endswith(".txt"):
42
+ fpath = os.path.join(BOOKS_DIR, fname)
43
+ if os.path.getsize(fpath) > 100:
44
+ files.append(fpath)
45
+ files.sort()
46
+ print(f"Found {len(files)} book files in {BOOKS_DIR}")
47
+ return files
48
+
49
+ def build_tokenizer():
50
+ # Create BPE tokenizer with ByteLevel pre-tokenizer
51
+ tokenizer = Tokenizer(models.BPE(unk_token="<unk>"))
52
+ tokenizer.pre_tokenizer = pre_tokenizers.ByteLevel(add_prefix_space=False)
53
+ tokenizer.decoder = decoders.ByteLevel(add_prefix_space=False)
54
+ tokenizer.add_special_tokens(special_tokens)
55
+
56
+ book_files = find_book_files()
57
+
58
+ print(f"Training BPE with vocab_size=4096, {len(book_files)} files...")
59
+ trainer = trainers.BpeTrainer(
60
+ vocab_size=4096,
61
+ min_frequency=2,
62
+ special_tokens=special_tokens,
63
+ show_progress=True,
64
+ initial_alphabet=[],
65
+ )
66
+ tokenizer.train(book_files, trainer)
67
+ print("BPE training done.")
68
+
69
+ output_path = os.path.join(TOKENIZER_DIR, "tokenizer.json")
70
+ tokenizer.save(output_path)
71
+ print(f"Saved to {output_path}")
72
+
73
+ # Post-process: ensure byte_fallback=True
74
+ with open(output_path) as f:
75
+ data = json.load(f)
76
+
77
+ data["model"]["byte_fallback"] = True
78
+ data["model"]["dropout"] = None
79
+
80
+ with open(output_path, "w") as f:
81
+ json.dump(data, f, ensure_ascii=False)
82
+
83
+ # Verify
84
+ with open(output_path) as f:
85
+ data = json.load(f)
86
+ vocab = data["model"]["vocab"]
87
+ merges = data["model"]["merges"]
88
+ sorted_vocab = sorted(vocab.items(), key=lambda x: x[1])
89
+ print(f"Vocab size: {len(vocab)}")
90
+ print(f"Merges: {len(merges)}")
91
+ print(f"First tokens: {sorted_vocab[:7]}")
92
+ print(f"Tokens 50-55: {sorted_vocab[50:55]}")
93
+ print(f"Last tokens: {sorted_vocab[-5:]}")
94
+ if merges:
95
+ print(f"First merges: {merges[:5]}")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
96
 
97
  if __name__ == "__main__":
98
+ build_tokenizer()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
infer_gguf.py CHANGED
@@ -1,47 +1,287 @@
1
- """
2
- Inference with GGUF INT4 model or QLoRA checkpoint.
3
-
4
- Usage:
5
- python3 scripts/infer_gguf.py --gguf outputs/tiny-sft/tiny.gguf
6
- python3 scripts/infer_gguf.py --checkpoint model.pt
7
- """
8
-
9
- import os, sys, argparse
10
- import gguf
11
  import torch
12
  import torch.nn.functional as F
13
 
14
  sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
15
- from scripts.model_tiny import TinyModel, apply_qlora
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
16
 
17
 
18
- def quantize_to_q4(tensor):
19
- t = tensor.float()
20
- max_val = t.abs().max()
21
- if max_val < 1e-8:
22
- return t
23
- scale = max_val / 7.0
24
- q = (t / scale).round().clamp(-7, 7).char()
25
- dq = q.float() * scale
26
- return dq
27
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
28
 
29
- def load_gguf_model(gguf_path):
 
 
 
30
  print(f" Loading GGUF: {gguf_path}")
31
- reader = gguf.GGUFReader(gguf_path)
 
32
 
33
- model = TinyModel(
34
- vocab_size=1757, hidden=128, intermediate=640,
35
- num_layers=3, num_heads=8, num_kv_heads=4,
36
- max_seq_len=2048, tie_weights=True,
37
- )
38
  model.eval()
39
 
40
  state = {}
41
  for tensor in reader.tensors:
42
  name = tensor.name
43
  data = torch.from_numpy(tensor.data.copy())
44
-
45
  if name == "token_embd.weight":
46
  state["token_embed.weight"] = data
47
  elif name == "output_norm.weight":
@@ -73,56 +313,39 @@ def load_gguf_model(gguf_path):
73
 
74
  model.load_state_dict(state, strict=False)
75
  print(f" Loaded: {len(state)} tensors")
76
-
77
- for name, param in model.named_parameters():
78
- if "weight" in name and "norm" not in name and "embed" not in name:
79
- param.data.copy_(quantize_to_q4(param.data))
80
-
81
  return model
82
 
83
 
84
- def load_checkpoint_model(ckpt_path):
85
  print(f" Loading checkpoint: {ckpt_path}")
86
  state = torch.load(ckpt_path, map_location="cpu", weights_only=True)
87
- has_qlora = any("qweight" in k for k in state)
88
- print(f" Detected: {'QLoRA' if has_qlora else 'full'} checkpoint")
89
-
90
- model = TinyModel(
91
- vocab_size=1757, hidden=128, intermediate=640,
92
- num_layers=3, num_heads=8, num_kv_heads=4,
93
- max_seq_len=2048, tie_weights=True,
94
- )
95
-
96
- if has_qlora:
97
- model = apply_qlora(model,
98
- target_modules=["q_proj","k_proj","v_proj","o_proj","gate","up","down"],
99
- r=8, alpha=16, dropout=0.0, freeze_embeds=True)
100
-
101
- missing, unexpected = model.load_state_dict(state, strict=False)
102
- if missing:
103
- print(f" Missing keys: {missing}")
104
- if unexpected:
105
- print(f" Unexpected keys: {unexpected}")
106
  model.eval()
107
- return model
108
-
109
 
110
- def load_tokenizer():
111
- from tokenizers import Tokenizer as Tk
112
- tok = Tk.from_file(os.path.join(os.path.dirname(__file__), "..", "tokenizer", "tokenizer.json"))
113
- return tok
114
-
115
-
116
- def _build_chat(system, user, tokenizer):
117
- parts = [f"<|system|>\n{system}", f"<|user|>\n{user}", "<|assistant|>\n"]
118
- text = "\n".join(parts)
119
- return tokenizer.encode(text).ids
120
 
 
121
 
122
  def main():
123
- parser = argparse.ArgumentParser()
124
- parser.add_argument("--gguf", default=None)
125
- parser.add_argument("--checkpoint", default=None)
126
  parser.add_argument("--prompt", default="What is 2+2?")
127
  parser.add_argument("--system", default="You are a helpful AI assistant.")
128
  parser.add_argument("--max-tokens", type=int, default=128)
@@ -135,16 +358,18 @@ def main():
135
  sys.exit(1)
136
 
137
  model = None
 
138
  if args.gguf:
139
  if not os.path.exists(args.gguf):
140
  print(f"GGUF not found: {args.gguf}")
141
  sys.exit(1)
142
- model = load_gguf_model(args.gguf)
 
143
  elif args.checkpoint:
144
  if not os.path.exists(args.checkpoint):
145
  print(f"Checkpoint not found: {args.checkpoint}")
146
  sys.exit(1)
147
- model = load_checkpoint_model(args.checkpoint)
148
  else:
149
  print("Specify --gguf or --checkpoint")
150
  sys.exit(1)
@@ -153,17 +378,16 @@ def main():
153
  model.to(device)
154
  print(f" Device: {device}")
155
 
156
- tok = load_tokenizer()
157
- input_ids = _build_chat(args.system, args.prompt, tok)
158
  input_ids = torch.tensor([input_ids], device=device)
159
 
160
  print("\n" + tok.decode(input_ids[0].tolist()), end="", flush=True)
161
 
162
  def stream(id_):
163
- t = tok.decode([id_])
164
- print(t, end="", flush=True)
165
 
166
- out = model.generate(
167
  input_ids,
168
  max_new_tokens=args.max_tokens,
169
  temperature=args.temperature,
 
1
+ import os, sys, argparse, math as _math
 
 
 
 
 
 
 
 
 
2
  import torch
3
  import torch.nn.functional as F
4
 
5
  sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
6
+
7
+ HERE = os.path.dirname(os.path.abspath(__file__))
8
+ PROJ = os.path.dirname(HERE)
9
+
10
+
11
+ # ─── V2 model (backward compat: GGUF + QLoRA checkpoints) ───────────────
12
+
13
+ class ALiBiAttention(torch.nn.Module):
14
+ def __init__(self, hidden: int, num_heads: int, num_kv_heads: int):
15
+ super().__init__()
16
+ self.num_heads = num_heads
17
+ self.num_kv_heads = num_kv_heads
18
+ self.head_dim = hidden // num_heads
19
+ self.num_groups = num_heads // num_kv_heads
20
+ self.q_proj = torch.nn.Linear(hidden, hidden, bias=False)
21
+ self.k_proj = torch.nn.Linear(hidden, num_kv_heads * self.head_dim, bias=False)
22
+ self.v_proj = torch.nn.Linear(hidden, num_kv_heads * self.head_dim, bias=False)
23
+ self.o_proj = torch.nn.Linear(hidden, hidden, bias=False)
24
+
25
+ @staticmethod
26
+ def _get_alibi_slopes(num_heads):
27
+ n = 2 ** _math.ceil(_math.log2(num_heads))
28
+ return torch.tensor([2.0 ** (-(i + 1)) for i in range(n)][:num_heads])
29
+
30
+ def forward(self, x):
31
+ B, T, C = x.shape
32
+ q = self.q_proj(x).view(B, T, self.num_heads, self.head_dim).transpose(1, 2)
33
+ k = self.k_proj(x).view(B, T, self.num_kv_heads, self.head_dim).transpose(1, 2)
34
+ v = self.v_proj(x).view(B, T, self.num_kv_heads, self.head_dim).transpose(1, 2)
35
+ k = k.repeat_interleave(self.num_groups, dim=1)
36
+ v = v.repeat_interleave(self.num_groups, dim=1)
37
+ scores = (q @ k.transpose(-2, -1)) * (self.head_dim ** -0.5)
38
+ slopes = self._get_alibi_slopes(self.num_heads).to(x.device, x.dtype)
39
+ pos = torch.arange(T, device=x.device)
40
+ alibi = (pos.view(1, T) - pos.view(T, 1)).abs().neg().unsqueeze(0).unsqueeze(0)
41
+ scores = scores + alibi * slopes.view(-1, 1, 1)
42
+ causal = torch.triu(torch.full((T, T), float('-inf'), device=x.device, dtype=x.dtype), diagonal=1)
43
+ scores = scores + causal
44
+ attn = F.softmax(scores, dim=-1, dtype=torch.float32).to(x.dtype)
45
+ out = (attn @ v).transpose(1, 2).contiguous().view(B, T, C)
46
+ return self.o_proj(out)
47
+
48
+
49
+ class RMSNormV2(torch.nn.Module):
50
+ def __init__(self, dim: int, eps: float = 1e-6):
51
+ super().__init__()
52
+ self.weight = torch.nn.Parameter(torch.ones(dim))
53
+ self.eps = eps
54
+
55
+ def forward(self, x):
56
+ norm = x.float().pow(2).mean(-1, keepdim=True).add(self.eps).rsqrt()
57
+ return (x.float() * norm).type_as(x) * self.weight
58
+
59
+
60
+ class SwiGLU(torch.nn.Module):
61
+ def __init__(self, hidden: int, intermediate: int):
62
+ super().__init__()
63
+ self.gate = torch.nn.Linear(hidden, intermediate, bias=False)
64
+ self.up = torch.nn.Linear(hidden, intermediate, bias=False)
65
+ self.down = torch.nn.Linear(intermediate, hidden, bias=False)
66
+
67
+ def forward(self, x):
68
+ return self.down(F.silu(self.gate(x)) * self.up(x))
69
+
70
+
71
+ class TransformerBlockV2(torch.nn.Module):
72
+ def __init__(self, hidden: int, intermediate: int, num_heads: int, num_kv_heads: int):
73
+ super().__init__()
74
+ self.ln1 = RMSNormV2(hidden)
75
+ self.attn = ALiBiAttention(hidden, num_heads, num_kv_heads)
76
+ self.ln2 = RMSNormV2(hidden)
77
+ self.mlp = SwiGLU(hidden, intermediate)
78
+
79
+ def forward(self, x):
80
+ x = x + self.attn(self.ln1(x))
81
+ x = x + self.mlp(self.ln2(x))
82
+ return x
83
+
84
+
85
+ class TinyModelV2(torch.nn.Module):
86
+ def __init__(self, vocab_size=1757, hidden=128, intermediate=640,
87
+ num_layers=3, num_heads=8, num_kv_heads=4, max_seq_len=2048,
88
+ tie_weights=True):
89
+ super().__init__()
90
+ self.max_seq_len = max_seq_len
91
+ self.token_embed = torch.nn.Embedding(vocab_size, hidden)
92
+ self.blocks = torch.nn.ModuleList([
93
+ TransformerBlockV2(hidden, intermediate, num_heads, num_kv_heads)
94
+ for _ in range(num_layers)
95
+ ])
96
+ self.ln_f = RMSNormV2(hidden)
97
+ self.lm_head = torch.nn.Linear(hidden, vocab_size, bias=False)
98
+ if tie_weights:
99
+ self.lm_head.weight = self.token_embed.weight
100
+
101
+ def forward(self, input_ids):
102
+ x = self.token_embed(input_ids)
103
+ for block in self.blocks:
104
+ x = block(x)
105
+ x = self.ln_f(x)
106
+ return self.lm_head(x)
107
+
108
+ @torch.no_grad()
109
+ def generate(self, input_ids, max_new_tokens=128, temperature=0.7, top_p=0.9, stream_callback=None):
110
+ self.eval()
111
+ for _ in range(max_new_tokens):
112
+ if input_ids.size(1) > self.max_seq_len:
113
+ input_ids = input_ids[:, -self.max_seq_len:]
114
+ logits = self.forward(input_ids)
115
+ logits = logits[:, -1, :]
116
+ if temperature > 0:
117
+ logits = logits / temperature
118
+ if top_p < 1.0:
119
+ sorted_logits, sorted_idx = logits.sort(dim=-1, descending=True)
120
+ cum_probs = sorted_logits.softmax(dim=-1).cumsum(dim=-1)
121
+ cutoff = cum_probs > top_p
122
+ cutoff[..., 1:] = cutoff[..., :-1].clone()
123
+ cutoff[..., 0] = False
124
+ logits[~cutoff] = float('-inf')
125
+ probs = F.softmax(logits, dim=-1)
126
+ if temperature > 0:
127
+ next_token = torch.multinomial(probs, 1)
128
+ else:
129
+ next_token = probs.argmax(dim=-1, keepdim=True)
130
+ input_ids = torch.cat([input_ids, next_token], dim=1)
131
+ if stream_callback:
132
+ stream_callback(next_token.item())
133
+ if next_token.item() == 2:
134
+ break
135
+ return input_ids
136
+
137
+
138
+ # ─── QLoRA (NF4 + LoRA) for V2 ─────────────────────────────────────────
139
+
140
+ NF4_LEVELS = torch.tensor([
141
+ -1.0, -0.6961928009986877, -0.5250730514526367, -0.39491748809814453,
142
+ -0.28444138169288635, -0.18477343022823334, -0.09105003625154495, 0.0,
143
+ 0.07958029955625534, 0.16093020141124725, 0.24611230194568634,
144
+ 0.33791524171829224, 0.44070982933044434, 0.5626170039176941,
145
+ 0.7229568362236023, 1.0,
146
+ ])
147
+
148
+ def unpack_nf4(qweight, shape):
149
+ n = qweight.numel() * 2
150
+ lo = (qweight & 0x0F).view(-1)
151
+ hi = ((qweight >> 4) & 0x0F).view(-1)
152
+ indices = torch.stack([lo, hi], dim=1).reshape(n)
153
+ return indices[:shape[0] * shape[1]].reshape(shape)
154
+
155
+ def pack_nf4(indices):
156
+ n = indices.numel()
157
+ even = indices[0::2]
158
+ odd = indices[1::2]
159
+ packed = (odd << 4) | even
160
+ return packed.to(torch.uint8)
161
+
162
+
163
+ class QLoRALinear(torch.nn.Module):
164
+ def __init__(self, in_features: int, out_features: int, r: int = 8, alpha: float = 16, dropout: float = 0.0):
165
+ super().__init__()
166
+ self.in_features = in_features
167
+ self.out_features = out_features
168
+ self.r = r
169
+ self.scaling = alpha / r
170
+ self.dropout = torch.nn.Dropout(dropout)
171
+ self.register_buffer("qweight", torch.zeros((in_features * out_features + 1) // 2, dtype=torch.uint8))
172
+ self.register_buffer("scales", torch.zeros(out_features))
173
+ self.register_buffer("bias", torch.zeros(out_features))
174
+ self.lora_A = torch.nn.Parameter(torch.zeros(r, in_features))
175
+ self.lora_B = torch.nn.Parameter(torch.zeros(out_features, r))
176
+ torch.nn.init.kaiming_uniform_(self.lora_A, a=_math.sqrt(5))
177
+
178
+ def dequantize(self):
179
+ indices = unpack_nf4(self.qweight, (self.out_features, self.in_features))
180
+ return NF4_LEVELS.to(self.qweight.device)[indices.long()] * self.scales[:, None]
181
+
182
+ def quantize_from(self, weight, bias=None):
183
+ w = weight.float()
184
+ scales = w.abs().max(dim=1, keepdim=True).values.clamp(min=1e-8)
185
+ scaled = (w / scales).clamp(-1, 1)
186
+ idx = (scaled[:, :, None] - NF4_LEVELS[None, None, :].to(w.device)).abs().argmin(dim=-1)
187
+ self.qweight.copy_(pack_nf4(idx.reshape(-1)))
188
+ self.scales.copy_(scales.squeeze(1))
189
+ if bias is not None:
190
+ self.bias.copy_(bias.float())
191
+
192
+ def forward(self, x):
193
+ base = F.linear(x, self.dequantize(), self.bias)
194
+ return base + self.dropout(x) @ self.lora_A.T @ self.lora_B.T * self.scaling
195
+
196
+
197
+ def apply_qlora_v2(model, target_modules=None, r=8, alpha=16, dropout=0.0):
198
+ if target_modules is None:
199
+ target_modules = ["q_proj", "k_proj", "v_proj", "o_proj", "gate", "up", "down"]
200
+ qlora_params = 0
201
+ for name, module in model.named_modules():
202
+ if not isinstance(module, torch.nn.Linear):
203
+ continue
204
+ key = name.split(".")[-1]
205
+ if key not in target_modules:
206
+ continue
207
+ parent = model
208
+ parts = name.split(".")
209
+ for p in parts[:-1]:
210
+ parent = getattr(parent, p)
211
+ qlora = QLoRALinear(module.in_features, module.out_features, r=r, alpha=alpha, dropout=dropout)
212
+ qlora.quantize_from(module.weight, module.bias)
213
+ setattr(parent, parts[-1], qlora)
214
+ qlora_params += 2 * r * module.in_features + module.out_features * r
215
+ n = sum(p.numel() for p in model.parameters() if p.requires_grad)
216
+ print(f" QLoRA applied: {qlora_params:,} LoRA params | trainable: {n:,}")
217
+
218
+
219
+ # ─── V3 model (from model_tiny) ─────────────────────────────────────────
220
+
221
+ def _load_v3_model():
222
+ from scripts.model_tiny import TinyModel as TinyModelV3
223
+ return TinyModelV3
224
+
225
+
226
+ # ─── Tokenizer ──────────────────────────────────────────────────────────
227
+
228
+ def load_tokenizer(ckpt_type=None):
229
+ from tokenizers import Tokenizer as Tk
230
+ paths = [os.path.join(PROJ, "tokenizer", "tokenizer.json"),
231
+ os.path.join(PROJ, "tokenizer.json")]
232
+ if ckpt_type and ckpt_type.startswith("v2"):
233
+ v2_path = os.path.join(os.path.dirname(PROJ), "lumia-v1", "tokenizer", "tokenizer.json")
234
+ if os.path.exists(v2_path):
235
+ paths.insert(0, v2_path)
236
+ for p in paths:
237
+ if os.path.exists(p):
238
+ tok = Tk.from_file(p)
239
+ print(f" Tokenizer: {p} ({tok.get_vocab_size()} vocab)")
240
+ return tok
241
+ print(" Tokenizer not found")
242
+ sys.exit(1)
243
+
244
+
245
+ def build_chat(system, user, tokenizer):
246
+ parts = [f"<|system|>\n{system}", f"<|user|>\n{user}", "<|assistant|>\n"]
247
+ text = "\n".join(parts)
248
+ return tokenizer.encode(text).ids
249
 
250
 
251
+ # ─── Detect checkpoint type ─────────────────────────────────────────────
 
 
 
 
 
 
 
 
252
 
253
+ def _detect_ckpt_type(state):
254
+ keys = list(state.keys())
255
+ if any("rpw" in k or "vcr" in k for k in keys):
256
+ return "v3"
257
+ if any("qweight" in k for k in keys):
258
+ return "v2_qlora"
259
+ if any("gate" in k for k in keys) or any("mlp.gate" in k for k in keys):
260
+ return "v2_fp32"
261
+ if any(k.startswith("blocks.") for k in keys) and any("ln1.weight" in k for k in keys):
262
+ blk_keys = [k.split(".")[1] for k in keys if k.startswith("blocks.")]
263
+ max_blk = max(int(b) for b in blk_keys) if blk_keys else 0
264
+ if max_blk >= 3:
265
+ return "v3"
266
+ return "v2_fp32"
267
 
268
+
269
+ # ─── Loaders ────────────────────────────────────────────────────────────
270
+
271
+ def load_gguf(gguf_path):
272
  print(f" Loading GGUF: {gguf_path}")
273
+ import gguf as _gguf
274
+ reader = _gguf.GGUFReader(gguf_path)
275
 
276
+ model = TinyModelV2(vocab_size=1757, hidden=128, intermediate=640,
277
+ num_layers=3, num_heads=8, num_kv_heads=4,
278
+ max_seq_len=2048, tie_weights=True)
 
 
279
  model.eval()
280
 
281
  state = {}
282
  for tensor in reader.tensors:
283
  name = tensor.name
284
  data = torch.from_numpy(tensor.data.copy())
 
285
  if name == "token_embd.weight":
286
  state["token_embed.weight"] = data
287
  elif name == "output_norm.weight":
 
313
 
314
  model.load_state_dict(state, strict=False)
315
  print(f" Loaded: {len(state)} tensors")
 
 
 
 
 
316
  return model
317
 
318
 
319
+ def load_checkpoint(ckpt_path):
320
  print(f" Loading checkpoint: {ckpt_path}")
321
  state = torch.load(ckpt_path, map_location="cpu", weights_only=True)
322
+ ckpt_type = _detect_ckpt_type(state)
323
+ print(f" Detected: {ckpt_type}")
324
+
325
+ if ckpt_type.startswith("v2"):
326
+ has_qlora = ckpt_type == "v2_qlora"
327
+ model = TinyModelV2(vocab_size=1757, hidden=128, intermediate=640,
328
+ num_layers=3, num_heads=8, num_kv_heads=4,
329
+ max_seq_len=2048, tie_weights=True)
330
+ if has_qlora:
331
+ apply_qlora_v2(model)
332
+ model.load_state_dict(state, strict=False)
333
+ model.eval()
334
+ return model, ckpt_type
335
+
336
+ TV3 = _load_v3_model()
337
+ model = TV3()
338
+ model.load_state_dict(state)
 
 
339
  model.eval()
340
+ return model, ckpt_type
 
341
 
 
 
 
 
 
 
 
 
 
 
342
 
343
+ # ─── Main ───────────────────────────────────────────────────────────────
344
 
345
  def main():
346
+ parser = argparse.ArgumentParser(description="Infer Lumia V2 (GGUF/QLoRA) or V3 (best.pt)")
347
+ parser.add_argument("--gguf", default=None, help="V2 GGUF path")
348
+ parser.add_argument("--checkpoint", default="best.pt", help="Checkpoint path (default: best.pt)")
349
  parser.add_argument("--prompt", default="What is 2+2?")
350
  parser.add_argument("--system", default="You are a helpful AI assistant.")
351
  parser.add_argument("--max-tokens", type=int, default=128)
 
358
  sys.exit(1)
359
 
360
  model = None
361
+ ckpt_type = None
362
  if args.gguf:
363
  if not os.path.exists(args.gguf):
364
  print(f"GGUF not found: {args.gguf}")
365
  sys.exit(1)
366
+ model = load_gguf(args.gguf)
367
+ ckpt_type = "gguf"
368
  elif args.checkpoint:
369
  if not os.path.exists(args.checkpoint):
370
  print(f"Checkpoint not found: {args.checkpoint}")
371
  sys.exit(1)
372
+ model, ckpt_type = load_checkpoint(args.checkpoint)
373
  else:
374
  print("Specify --gguf or --checkpoint")
375
  sys.exit(1)
 
378
  model.to(device)
379
  print(f" Device: {device}")
380
 
381
+ tok = load_tokenizer(ckpt_type)
382
+ input_ids = build_chat(args.system, args.prompt, tok)
383
  input_ids = torch.tensor([input_ids], device=device)
384
 
385
  print("\n" + tok.decode(input_ids[0].tolist()), end="", flush=True)
386
 
387
  def stream(id_):
388
+ print(tok.decode([id_]), end="", flush=True)
 
389
 
390
+ model.generate(
391
  input_ids,
392
  max_new_tokens=args.max_tokens,
393
  temperature=args.temperature,
model_tiny.py CHANGED
@@ -175,8 +175,9 @@ class TinyModel(nn.Module):
175
  input_ids = input_ids[:, -self.max_seq_len:]
176
  with torch.no_grad():
177
  logits, _ = self.forward(input_ids)
178
- logits = logits[:, -1, :] / temperature
179
-
 
180
  if top_p < 1.0:
181
  sorted_logits, sorted_idx = logits.sort(dim=-1, descending=True)
182
  cum_probs = sorted_logits.softmax(dim=-1).cumsum(dim=-1)
@@ -184,14 +185,14 @@ class TinyModel(nn.Module):
184
  cutoff[..., 1:] = cutoff[..., :-1].clone()
185
  cutoff[..., 0] = False
186
  logits[~cutoff] = float('-inf')
187
-
188
  probs = F.softmax(logits, dim=-1)
189
- next_token = torch.multinomial(probs, num_samples=1)
 
 
 
190
  input_ids = torch.cat([input_ids, next_token], dim=1)
191
-
192
  if stream_callback:
193
  stream_callback(next_token.item())
194
-
195
  if next_token.item() == 2:
196
  break
197
  return input_ids
@@ -244,6 +245,120 @@ def restore_from_v2(v2_ckpt_path: str | None = None, strict: bool = False) -> Ti
244
  return model
245
 
246
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
247
  # ─── Helpers ───────────────────────────────────────────────────────────────
248
 
249
  def count_params(model):
 
175
  input_ids = input_ids[:, -self.max_seq_len:]
176
  with torch.no_grad():
177
  logits, _ = self.forward(input_ids)
178
+ logits = logits[:, -1, :]
179
+ if temperature > 0:
180
+ logits = logits / temperature
181
  if top_p < 1.0:
182
  sorted_logits, sorted_idx = logits.sort(dim=-1, descending=True)
183
  cum_probs = sorted_logits.softmax(dim=-1).cumsum(dim=-1)
 
185
  cutoff[..., 1:] = cutoff[..., :-1].clone()
186
  cutoff[..., 0] = False
187
  logits[~cutoff] = float('-inf')
 
188
  probs = F.softmax(logits, dim=-1)
189
+ if temperature > 0:
190
+ next_token = torch.multinomial(probs, num_samples=1)
191
+ else:
192
+ next_token = probs.argmax(dim=-1, keepdim=True)
193
  input_ids = torch.cat([input_ids, next_token], dim=1)
 
194
  if stream_callback:
195
  stream_callback(next_token.item())
 
196
  if next_token.item() == 2:
197
  break
198
  return input_ids
 
245
  return model
246
 
247
 
248
+ # ─── QLoRA (4-bit NF4 + LoRA) ──────────────────────────────────────────────
249
+
250
+ NF4_LEVELS = torch.tensor([
251
+ -1.0, -0.6961928009986877, -0.5250730514526367, -0.39491748809814453,
252
+ -0.28444138169288635, -0.18477343022823334, -0.09105003625154495, 0.0,
253
+ 0.07958029955625534, 0.16093020141124725, 0.24611230194568634,
254
+ 0.33791524171829224, 0.44070982933044434, 0.5626170039176941,
255
+ 0.7229568362236023, 1.0,
256
+ ], dtype=torch.float32)
257
+
258
+
259
+ def _quantize_nf4_row(row: torch.Tensor) -> tuple:
260
+ absmax = row.abs().max().clamp(min=1e-12)
261
+ scaled = row / absmax
262
+ idx = (scaled[:, None] - NF4_LEVELS[None, :].to(row.device)).abs().argmin(dim=-1)
263
+ return idx.to(torch.uint8), absmax
264
+
265
+
266
+ def pack_nf4(indices: torch.Tensor) -> torch.Tensor:
267
+ n = indices.numel()
268
+ if n % 2 != 0:
269
+ indices = torch.cat([indices, indices.new_zeros(1)])
270
+ packed = (indices[0::2].to(torch.uint8) | (indices[1::2].to(torch.uint8) << 4))
271
+ return packed
272
+
273
+
274
+ def unpack_nf4(packed: torch.Tensor, shape) -> torch.Tensor:
275
+ n = shape[0] * shape[1]
276
+ low = (packed & 0x0F).to(torch.long)
277
+ high = ((packed >> 4) & 0x0F).to(torch.long)
278
+ indices = torch.stack([low, high], dim=-1).reshape(n)
279
+ return indices[:shape[0] * shape[1]].reshape(shape)
280
+
281
+
282
+ def dequantize_nf4(packed_weight: torch.Tensor, scales: torch.Tensor, shape) -> torch.Tensor:
283
+ indices = unpack_nf4(packed_weight, shape)
284
+ return NF4_LEVELS[indices] * scales[:, None]
285
+
286
+
287
+ class QLoRALinear(nn.Module):
288
+ def __init__(self, in_features: int, out_features: int, r: int = 8, alpha: float = 16, dropout: float = 0.0):
289
+ super().__init__()
290
+ self.in_features = in_features
291
+ self.out_features = out_features
292
+ self.r = r
293
+ self.alpha = alpha
294
+ self.scaling = alpha / r
295
+ self.dropout = nn.Dropout(dropout) if dropout > 0 else nn.Identity()
296
+
297
+ n_elements = in_features * out_features
298
+ n_packed = (n_elements + 1) // 2
299
+ self.register_buffer("qweight", torch.zeros(n_packed, dtype=torch.uint8))
300
+ self.register_buffer("scales", torch.zeros(out_features, dtype=torch.float32))
301
+ self.register_buffer("bias", torch.zeros(out_features, dtype=torch.float32))
302
+ self._has_bias = False
303
+ self._shape = (out_features, in_features)
304
+
305
+ self.lora_A = nn.Parameter(torch.zeros(r, in_features))
306
+ self.lora_B = nn.Parameter(torch.zeros(out_features, r))
307
+ nn.init.kaiming_uniform_(self.lora_A, a=math.sqrt(5))
308
+
309
+ def _dequantized_weight(self) -> torch.Tensor:
310
+ dev = self.qweight.device
311
+ indices = unpack_nf4(self.qweight, self._shape)
312
+ return NF4_LEVELS.to(dev)[indices] * self.scales.to(dev)[:, None]
313
+
314
+ def quantize_from(self, weight: torch.Tensor, bias: torch.Tensor | None = None):
315
+ w = weight.float().detach()
316
+ rows = []
317
+ scales = []
318
+ for i in range(w.shape[0]):
319
+ idx, s = _quantize_nf4_row(w[i])
320
+ rows.append(idx)
321
+ scales.append(s.item())
322
+ all_idx = torch.stack(rows)
323
+ self.qweight.copy_(pack_nf4(all_idx.reshape(-1)))
324
+ self.scales.copy_(torch.tensor(scales, dtype=torch.float32))
325
+ if bias is not None:
326
+ self.bias.copy_(bias.float().detach())
327
+ self._has_bias = True
328
+
329
+ def forward(self, x):
330
+ w = self._dequantized_weight()
331
+ b = self.bias if self._has_bias else None
332
+ base = F.linear(x, w, b)
333
+ return base + self.dropout(x) @ self.lora_A.T @ self.lora_B.T * self.scaling
334
+
335
+
336
+ def apply_qlora(model, target_modules=None, r=8, alpha=16, dropout=0.0, freeze_embeds=True):
337
+ if target_modules is None:
338
+ target_modules = ["q_proj", "k_proj", "v_proj", "o_proj", "down", "up"]
339
+ qlora_params = 0
340
+ for name, module in model.named_modules():
341
+ if not isinstance(module, nn.Linear):
342
+ continue
343
+ key = name.split(".")[-1]
344
+ if key not in target_modules:
345
+ continue
346
+ parent = model
347
+ parts = name.split(".")
348
+ for p in parts[:-1]:
349
+ parent = getattr(parent, p)
350
+ qlora = QLoRALinear(module.in_features, module.out_features, r=r, alpha=alpha, dropout=dropout)
351
+ qlora.quantize_from(module.weight, module.bias)
352
+ setattr(parent, parts[-1], qlora)
353
+ qlora_params += 2 * r * module.in_features + module.out_features * r
354
+ for name, param in model.named_parameters():
355
+ if "lora_" not in name:
356
+ param.requires_grad = False
357
+ n = sum(p.numel() for p in model.parameters() if p.requires_grad)
358
+ print(f"QLoRA applied: {qlora_params:,} LoRA params | trainable: {n:,}")
359
+ return model
360
+
361
+
362
  # ─── Helpers ───────────────────────────────────────────────────────────────
363
 
364
  def count_params(model):
tokenizer.json CHANGED
The diff for this file is too large to render. See raw diff