summaryrefslogtreecommitdiff
path: root/tokenizer/byte/tokenizer.go
blob: 98982d0846a50035494b1ae69554b42ed3a473bb (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
package byte

import (
	"encoding/json"
	"io"

	"github.com/jonasknobloch/mbpe"
)

// type parameter for ID type in bpe tokenzier ?!

// Name? Byte is bad; it is a bijection but so are most if not all tokenizers; its the Alphabet?
// It would be compativle with GenericTokenizer[uint8] without pretokenization

// Tokenizer is a byte level tokenzier covering the all 2^8 bytes; Essentially each individual byte is mapped to a token ID.
// Serialization follows the standard vocab.json (with HF byte replacements) while omitting merges.txt
// Pre-tokenization is not necessary; however the tokenized strings can be much larger -> check allocations
type Tokenizer struct {
	atoi map[byte]uint8
	itoa map[uint8]byte
}

func (t *Tokenizer) Encode(s string) []byte {
	b := make([]byte, len(s))

	for i := range len(s) {
		v, ok := t.atoi[s[i]]

		if !ok {
			panic("unknown byte")
		}

		b[i] = v
	}

	return b
}

func (t *Tokenizer) Decode(ids []uint8) string {
	b := make([]byte, len(ids))

	for i, id := range ids {
		v, ok := t.itoa[id]

		if !ok {
			panic("unknown token ID")
		}

		b[i] = v
	}

	return string(b)
}

func (t *Tokenizer) Tokenize(s string) []int {
	ids := t.Encode(s)

	r := make([]int, len(ids))

	for i, id := range ids {
		r[i] = int(id)
	}

	return r
}

func NewTokenizer(vocab io.Reader) (*Tokenizer, error) {
	v := make(map[string]uint8)

	decoder := json.NewDecoder(vocab)

	if err := decoder.Decode(&v); err != nil {
		return nil, err
	}

	if len(v) != 256 {
		panic("vocabulary size != 256")
	}

	atoi := make(map[byte]uint8, len(v))
	itoa := make(map[uint8]byte, len(v))

	for char, id := range v {
		b := mbpe.CharBytes[char]

		atoi[b] = id
		itoa[id] = b
	}

	return &Tokenizer{
		atoi: atoi,
		itoa: itoa,
	}, nil
}