1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
|
package byte
import (
"encoding/json"
"io"
"github.com/jonasknobloch/mbpe"
)
// type parameter for ID type in bpe tokenzier ?!
// Name? Byte is bad; it is a bijection but so are most if not all tokenizers; its the Alphabet?
// It would be compativle with GenericTokenizer[uint8] without pretokenization
// Tokenizer is a byte level tokenzier covering the all 2^8 bytes; Essentially each individual byte is mapped to a token ID.
// Serialization follows the standard vocab.json (with HF byte replacements) while omitting merges.txt
// Pre-tokenization is not necessary; however the tokenized strings can be much larger -> check allocations
type Tokenizer struct {
atoi map[byte]uint8
itoa map[uint8]byte
}
func (t *Tokenizer) Encode(s string) []byte {
b := make([]byte, len(s))
for i := range len(s) {
v, ok := t.atoi[s[i]]
if !ok {
panic("unknown byte")
}
b[i] = v
}
return b
}
func (t *Tokenizer) Decode(ids []uint8) string {
b := make([]byte, len(ids))
for i, id := range ids {
v, ok := t.itoa[id]
if !ok {
panic("unknown token ID")
}
b[i] = v
}
return string(b)
}
func (t *Tokenizer) Tokenize(s string) []int {
ids := t.Encode(s)
r := make([]int, len(ids))
for i, id := range ids {
r[i] = int(id)
}
return r
}
func NewTokenizer(vocab io.Reader) (*Tokenizer, error) {
v := make(map[string]uint8)
decoder := json.NewDecoder(vocab)
if err := decoder.Decode(&v); err != nil {
return nil, err
}
if len(v) != 256 {
panic("vocabulary size != 256")
}
atoi := make(map[byte]uint8, len(v))
itoa := make(map[uint8]byte, len(v))
for char, id := range v {
b := mbpe.CharBytes[char]
atoi[b] = id
itoa[id] = b
}
return &Tokenizer{
atoi: atoi,
itoa: itoa,
}, nil
}
|