package knobloch // Tokenizer characterisation tables. Column meanings are documented in // FREQUENCIES.md; keep the two in sync. // // Terminology, used consistently in identifiers and column names: // // type a distinct pre-token, i.e. one dictionary entry // occurrence a pre-token weighted by its corpus count // token a subword unit, i.e. a vocabulary entry // // Morpheme labels attach to subword tokens, never to pre-token types, so the // morpheme columns of table B are named morph_* rather than types_*. // // Tokenizers are compared by token string, never by id: the same string carries // a different id in every vocabulary, so comparing ids would measure id // reassignment rather than segmentation. import ( "encoding/csv" "encoding/json" "fmt" "hash/fnv" "os" "slices" "sort" "strconv" "github.com/jonasknobloch/mbpe" "go.jknobloc.com/x/shelf" "go.jknobloc.com/x/tokenizer/bpe" ) // TableSpec identifies one tokenizer to profile. type TableSpec struct { Name string // tokenizer label, e.g. m050 Alignment string // alpha as a plain number; direction is carried by Inverted Inverted bool VocabSize int Dir shelf.Item // directory holding vocab.json and merges.txt } // Encoded is one tokenizer's view of the dictionary. // // Comparisons need to know, per dictionary entry, whether two tokenizers agree // and how many pieces each produced. Retaining the pieces themselves costs // hundreds of megabytes per tokenizer, which does not fit for a whole // vocabulary-size family at once, so each entry is reduced to a hash of its // segmentation plus two small counts. That is all tables A, B and C read. type Encoded struct { Spec TableSpec // corpus count per vocab id, and the id-to-string map Counts []int Itoa map[int64]string Words int64 Tokens int64 // pieces-per-type distribution, bucketed 1 / 2 / 3+ TypeBucket [3]int64 TokenBucket [3]int64 // per dictionary entry, in dict order Sig []uint64 // hash of the piece strings Len []uint16 // number of pieces MorphN []uint16 // number of pieces labelled morpheme } // Fertility is subword tokens per pre-token occurrence. func (e *Encoded) Fertility() float64 { if e.Words == 0 { return 0 } return float64(e.Tokens) / float64(e.Words) } // EncodeDict segments every dictionary entry with one tokenizer. func EncodeDict(spec TableSpec, items []mbpe.Chunk) (*Encoded, error) { morph, err := morphemes() if err != nil { return nil, err } tok, err := bpe.NewTokenizerFromFiles( shelf.Abs(spec.Dir+"/vocab.json"), shelf.Abs(spec.Dir+"/merges.txt"), bpe.Config{Recover: false}) if err != nil { return nil, err } // the dictionary is already pre-tokenised bpe.MBPE(tok).SetPreTokenizer(&NoPreTok{}) e := &Encoded{ Spec: spec, Counts: make([]int, len(bpe.Vocab(tok))), Itoa: bpe.Itoa(tok), Sig: make([]uint64, len(items)), Len: make([]uint16, len(items)), MorphN: make([]uint16, len(items)), } h := fnv.New64a() for i, v := range items { ids := tok.Encode(v.Src()) h.Reset() var morphN uint16 for _, id := range ids { s := e.Itoa[int64(id)] e.Counts[id] += v.N() // the separator keeps ab|c distinct from a|bc h.Write([]byte(s)) h.Write([]byte{0}) if _, ok := morph[s]; ok { morphN++ } } e.Sig[i] = h.Sum64() e.Len[i] = uint16(len(ids)) e.MorphN[i] = morphN n := int64(v.N()) k := int64(len(ids)) b := bucket(len(ids)) e.Words += n e.Tokens += n * k e.TypeBucket[b]++ e.TokenBucket[b] += n * k } return e, nil } // bucket maps a piece count onto the 1 / 2 / 3+ buckets used by every table. func bucket(k int) int { switch { case k == 1: return 0 case k == 2: return 1 default: return 2 } } // VocabIntersection returns the token strings present in every one of the given // vocabularies. Reads vocab.json only, so it never touches the corpus. func VocabIntersection(dirs []shelf.Item) (map[string]struct{}, error) { if len(dirs) == 0 { return nil, fmt.Errorf("no vocabularies given") } var keep map[string]struct{} for _, dir := range dirs { file, err := os.Open(shelf.Abs(dir + "/vocab.json")) if err != nil { return nil, err } var v map[string]int64 err = json.NewDecoder(file).Decode(&v) file.Close() if err != nil { return nil, err } if keep == nil { keep = make(map[string]struct{}, len(v)) for token := range v { keep[token] = struct{}{} } continue } for token := range keep { if _, ok := v[token]; !ok { delete(keep, token) } } } return keep, nil } // TableARow is the segmentation profile of one tokenizer. type TableARow struct { Spec TableSpec Fertility float64 Types [3]float64 // share of pre-token types Tokens [3]float64 // share of subword token occurrences } func BuildTableA(e *Encoded) TableARow { row := TableARow{Spec: e.Spec, Fertility: e.Fertility()} var typeTotal, tokenTotal int64 for b := range e.TypeBucket { typeTotal += e.TypeBucket[b] tokenTotal += e.TokenBucket[b] } for b := range e.TypeBucket { row.Types[b] = share(e.TypeBucket[b], typeTotal) row.Tokens[b] = share(e.TokenBucket[b], tokenTotal) } return row } // TableBRow describes what a vocabulary is made of. type TableBRow struct { Spec TableSpec Fertility float64 MorphCount int MorphShareUnweighted float64 MorphShareWeighted float64 PctRankMorph float64 PctRankOther float64 SharedMorph float64 SharedOther float64 } // BuildTableB profiles one vocabulary. shared is the intersection of every // vocabulary at this size, which is what the two shared_* columns measure // against — so the baseline is just another tokenizer and gets a real value. func BuildTableB(e *Encoded, shared map[string]struct{}) (TableBRow, error) { morph, err := morphemes() if err != nil { return TableBRow{}, err } row := TableBRow{Spec: e.Spec, Fertility: e.Fertility()} var morphTokens, otherTokens int64 var morphFreq, otherFreq []int var morphShared, otherShared, otherCount int for id, f := range e.Counts { token := e.Itoa[int64(id)] _, isMorph := morph[token] _, inShared := shared[token] if isMorph { row.MorphCount++ morphTokens += int64(f) morphFreq = append(morphFreq, f) if inShared { morphShared++ } } else { otherCount++ otherTokens += int64(f) otherFreq = append(otherFreq, f) if inShared { otherShared++ } } } row.MorphShareUnweighted = share(int64(row.MorphCount), int64(len(e.Counts))) row.MorphShareWeighted = share(morphTokens, morphTokens+otherTokens) // Percentile rank of each group's median token within the corpus frequency // distribution of the whole vocabulary. An absolute median is not comparable // across vocabulary sizes, since a given token is rarer relative to a larger // vocabulary; a percentile rank is. all := slices.Clone(e.Counts) slices.Sort(all) slices.Sort(morphFreq) slices.Sort(otherFreq) row.PctRankMorph = percentileRank(all, median(morphFreq)) row.PctRankOther = percentileRank(all, median(otherFreq)) row.SharedMorph = share(int64(morphShared), int64(row.MorphCount)) row.SharedOther = share(int64(otherShared), int64(otherCount)) return row, nil } // percentileRank is the share of sorted strictly below v, so a higher value // means a more frequent token. func percentileRank(sorted []int, v int) float64 { if len(sorted) == 0 { return 0 } return 100 * float64(sort.SearchInts(sorted, v)) / float64(len(sorted)) } // TableCRow is the divergence between one tokenizer and a reference. // // Everything is measured against the BASELINE segmentation. A pre-token is // bucketed by how many pieces the baseline produced for it and contributes that // many token occurrences, so a pre-token that is one token under the baseline // and two under the variant lands in bucket 1 and counts once. That makes the // three bucket columns sum to TokensDiff by construction, and keeps the buckets // the same population as the baseline's table A row. type TableCRow struct { Spec TableSpec Baseline string TypesDiff float64 TokensDiff float64 TokensDiffBucket [3]float64 // morpheme share of the subword tokens in the differing slice, under each // side, each normalised by that side's own token count DiffMorphBase float64 DiffMorphVar float64 } func BuildTableC(e, base *Encoded, items []mbpe.Chunk) (TableCRow, error) { if len(e.Sig) != len(base.Sig) { return TableCRow{}, fmt.Errorf("dictionaries differ in length") } row := TableCRow{Spec: e.Spec, Baseline: base.Spec.Name} var typDiff, diffTokens int64 var diffBucket [3]int64 var mBase, nBase, mVar, nVar int64 for i := range base.Sig { if e.Sig[i] == base.Sig[i] { continue } n := int64(items[i].N()) k := int64(base.Len[i]) typDiff++ diffTokens += n * k diffBucket[bucket(int(base.Len[i]))] += n * k mBase += n * int64(base.MorphN[i]) nBase += n * k mVar += n * int64(e.MorphN[i]) nVar += n * int64(e.Len[i]) } row.TypesDiff = share(typDiff, int64(len(base.Sig))) row.TokensDiff = share(diffTokens, base.Tokens) for b := range diffBucket { row.TokensDiffBucket[b] = share(diffBucket[b], base.Tokens) } row.DiffMorphBase = share(mBase, nBase) row.DiffMorphVar = share(mVar, nVar) return row, nil } func share(x, total int64) float64 { if total == 0 { return 0 } return 100 * float64(x) / float64(total) } func writeCSV(name string, header []string, rows [][]string) error { file, err := os.Create(name) if err != nil { return err } defer file.Close() w := csv.NewWriter(file) if err := w.Write(header); err != nil { return err } for _, r := range rows { if err := w.Write(r); err != nil { return err } } w.Flush() return w.Error() } // percentages are plain numbers, without a percent sign func f2(v float64) string { return strconv.FormatFloat(v, 'f', 2, 64) } func f4(v float64) string { return strconv.FormatFloat(v, 'f', 4, 64) } func WriteTableA(name string, rows []TableARow) error { out := make([][]string, 0, len(rows)) for _, r := range rows { out = append(out, []string{ r.Spec.Name, strconv.Itoa(r.Spec.VocabSize), r.Spec.Alignment, f4(r.Fertility), f2(r.Types[0]), f2(r.Types[1]), f2(r.Types[2]), f2(r.Tokens[0]), f2(r.Tokens[1]), f2(r.Tokens[2]), }) } return writeCSV(name, []string{ "tokenizer", "vocab_size", "alignment", "fertility", "types_1", "types_2", "types_3plus", "tokens_1", "tokens_2", "tokens_3plus", }, out) } func WriteTableB(name string, rows []TableBRow) error { out := make([][]string, 0, len(rows)) for _, r := range rows { out = append(out, []string{ r.Spec.Name, strconv.Itoa(r.Spec.VocabSize), r.Spec.Alignment, f4(r.Fertility), strconv.Itoa(r.MorphCount), f2(r.MorphShareUnweighted), f2(r.MorphShareWeighted), f2(r.PctRankMorph), f2(r.PctRankOther), f2(r.SharedMorph), f2(r.SharedOther), }) } return writeCSV(name, []string{ "tokenizer", "vocab_size", "alignment", "fertility", "morph_count", "morph_share_unweighted", "morph_share_weighted", "pct_rank_morph", "pct_rank_other", "shared_morph", "shared_other", }, out) } func WriteTableC(name string, rows []TableCRow) error { out := make([][]string, 0, len(rows)) for _, r := range rows { out = append(out, []string{ r.Spec.Name, strconv.Itoa(r.Spec.VocabSize), r.Spec.Alignment, r.Baseline, f2(r.TypesDiff), f2(r.TokensDiff), f2(r.TokensDiffBucket[0]), f2(r.TokensDiffBucket[1]), f2(r.TokensDiffBucket[2]), f2(r.DiffMorphBase), f2(r.DiffMorphVar), }) } return writeCSV(name, []string{ "tokenizer", "vocab_size", "alignment", "baseline", "types_diff", "tokens_diff", "tokens_1_diff", "tokens_2_diff", "tokens_3plus_diff", "tokens_diff_morph_base", "tokens_diff_morph_var", }, out) }