diff options
| author | Jonas Knobloch <jonas.knobloch@t-online.de> | 2026-09-11 18:39:02 +0200 |
|---|---|---|
| committer | Jonas Knobloch <jonas.knobloch@t-online.de> | 2026-09-11 18:39:02 +0200 |
| commit | 9e1b8c4bde9b0263a4c4d2278e3c283ca0eb07f1 (patch) | |
| tree | bf6b604a150a196490b92f89ca40a335aca83d31 /research/knobloch/tables.go | |
| parent | 75e581b3bc19a73d0c1f48b32f2e58121c9741c1 (diff) | |
Diffstat (limited to 'research/knobloch/tables.go')
| -rw-r--r-- | research/knobloch/tables.go | 483 |
1 files changed, 483 insertions, 0 deletions
diff --git a/research/knobloch/tables.go b/research/knobloch/tables.go new file mode 100644 index 0000000..c5b2dcf --- /dev/null +++ b/research/knobloch/tables.go @@ -0,0 +1,483 @@ +package knobloch + +// Tokenizer characterisation tables. Column meanings are documented in +// FREQUENCIES.md; keep the two in sync. +// +// Terminology, used consistently in identifiers and column names: +// +// type a distinct pre-token, i.e. one dictionary entry +// occurrence a pre-token weighted by its corpus count +// token a subword unit, i.e. a vocabulary entry +// +// Morpheme labels attach to subword tokens, never to pre-token types, so the +// morpheme columns of table B are named morph_* rather than types_*. +// +// Tokenizers are compared by token string, never by id: the same string carries +// a different id in every vocabulary, so comparing ids would measure id +// reassignment rather than segmentation. + +import ( + "encoding/csv" + "encoding/json" + "fmt" + "hash/fnv" + "os" + "slices" + "sort" + "strconv" + + "github.com/jonasknobloch/mbpe" + "go.jknobloc.com/x/shelf" + "go.jknobloc.com/x/tokenizer/bpe" +) + +// TableSpec identifies one tokenizer to profile. +type TableSpec struct { + Name string // tokenizer label, e.g. m050 + Alignment string // alpha as a plain number; direction is carried by Inverted + Inverted bool + VocabSize int + Dir shelf.Item // directory holding vocab.json and merges.txt +} + +// Encoded is one tokenizer's view of the dictionary. +// +// Comparisons need to know, per dictionary entry, whether two tokenizers agree +// and how many pieces each produced. Retaining the pieces themselves costs +// hundreds of megabytes per tokenizer, which does not fit for a whole +// vocabulary-size family at once, so each entry is reduced to a hash of its +// segmentation plus two small counts. That is all tables A, B and C read. +type Encoded struct { + Spec TableSpec + + // corpus count per vocab id, and the id-to-string map + Counts []int + Itoa map[int64]string + + Words int64 + Tokens int64 + + // pieces-per-type distribution, bucketed 1 / 2 / 3+ + TypeBucket [3]int64 + TokenBucket [3]int64 + + // per dictionary entry, in dict order + Sig []uint64 // hash of the piece strings + Len []uint16 // number of pieces + MorphN []uint16 // number of pieces labelled morpheme +} + +// Fertility is subword tokens per pre-token occurrence. +func (e *Encoded) Fertility() float64 { + if e.Words == 0 { + return 0 + } + + return float64(e.Tokens) / float64(e.Words) +} + +// EncodeDict segments every dictionary entry with one tokenizer. +func EncodeDict(spec TableSpec, items []mbpe.Chunk) (*Encoded, error) { + morph, err := morphemes() + + if err != nil { + return nil, err + } + + tok, err := bpe.NewTokenizerFromFiles( + shelf.Abs(spec.Dir+"/vocab.json"), shelf.Abs(spec.Dir+"/merges.txt"), + bpe.Config{Recover: false}) + + if err != nil { + return nil, err + } + + // the dictionary is already pre-tokenised + bpe.MBPE(tok).SetPreTokenizer(&NoPreTok{}) + + e := &Encoded{ + Spec: spec, + Counts: make([]int, len(bpe.Vocab(tok))), + Itoa: bpe.Itoa(tok), + Sig: make([]uint64, len(items)), + Len: make([]uint16, len(items)), + MorphN: make([]uint16, len(items)), + } + + h := fnv.New64a() + + for i, v := range items { + ids := tok.Encode(v.Src()) + + h.Reset() + + var morphN uint16 + + for _, id := range ids { + s := e.Itoa[int64(id)] + + e.Counts[id] += v.N() + + // the separator keeps ab|c distinct from a|bc + h.Write([]byte(s)) + h.Write([]byte{0}) + + if _, ok := morph[s]; ok { + morphN++ + } + } + + e.Sig[i] = h.Sum64() + e.Len[i] = uint16(len(ids)) + e.MorphN[i] = morphN + + n := int64(v.N()) + k := int64(len(ids)) + b := bucket(len(ids)) + + e.Words += n + e.Tokens += n * k + + e.TypeBucket[b]++ + e.TokenBucket[b] += n * k + } + + return e, nil +} + +// bucket maps a piece count onto the 1 / 2 / 3+ buckets used by every table. +func bucket(k int) int { + switch { + case k == 1: + return 0 + case k == 2: + return 1 + default: + return 2 + } +} + +// VocabIntersection returns the token strings present in every one of the given +// vocabularies. Reads vocab.json only, so it never touches the corpus. +func VocabIntersection(dirs []shelf.Item) (map[string]struct{}, error) { + if len(dirs) == 0 { + return nil, fmt.Errorf("no vocabularies given") + } + + var keep map[string]struct{} + + for _, dir := range dirs { + file, err := os.Open(shelf.Abs(dir + "/vocab.json")) + + if err != nil { + return nil, err + } + + var v map[string]int64 + + err = json.NewDecoder(file).Decode(&v) + + file.Close() + + if err != nil { + return nil, err + } + + if keep == nil { + keep = make(map[string]struct{}, len(v)) + + for token := range v { + keep[token] = struct{}{} + } + + continue + } + + for token := range keep { + if _, ok := v[token]; !ok { + delete(keep, token) + } + } + } + + return keep, nil +} + +// TableARow is the segmentation profile of one tokenizer. +type TableARow struct { + Spec TableSpec + Fertility float64 + + Types [3]float64 // share of pre-token types + Tokens [3]float64 // share of subword token occurrences +} + +func BuildTableA(e *Encoded) TableARow { + row := TableARow{Spec: e.Spec, Fertility: e.Fertility()} + + var typeTotal, tokenTotal int64 + + for b := range e.TypeBucket { + typeTotal += e.TypeBucket[b] + tokenTotal += e.TokenBucket[b] + } + + for b := range e.TypeBucket { + row.Types[b] = share(e.TypeBucket[b], typeTotal) + row.Tokens[b] = share(e.TokenBucket[b], tokenTotal) + } + + return row +} + +// TableBRow describes what a vocabulary is made of. +type TableBRow struct { + Spec TableSpec + Fertility float64 + + MorphCount int + MorphShareUnweighted float64 + MorphShareWeighted float64 + + PctRankMorph float64 + PctRankOther float64 + + SharedMorph float64 + SharedOther float64 +} + +// BuildTableB profiles one vocabulary. shared is the intersection of every +// vocabulary at this size, which is what the two shared_* columns measure +// against — so the baseline is just another tokenizer and gets a real value. +func BuildTableB(e *Encoded, shared map[string]struct{}) (TableBRow, error) { + morph, err := morphemes() + + if err != nil { + return TableBRow{}, err + } + + row := TableBRow{Spec: e.Spec, Fertility: e.Fertility()} + + var morphTokens, otherTokens int64 + var morphFreq, otherFreq []int + var morphShared, otherShared, otherCount int + + for id, f := range e.Counts { + token := e.Itoa[int64(id)] + + _, isMorph := morph[token] + _, inShared := shared[token] + + if isMorph { + row.MorphCount++ + morphTokens += int64(f) + morphFreq = append(morphFreq, f) + + if inShared { + morphShared++ + } + } else { + otherCount++ + otherTokens += int64(f) + otherFreq = append(otherFreq, f) + + if inShared { + otherShared++ + } + } + } + + row.MorphShareUnweighted = share(int64(row.MorphCount), int64(len(e.Counts))) + row.MorphShareWeighted = share(morphTokens, morphTokens+otherTokens) + + // Percentile rank of each group's median token within the corpus frequency + // distribution of the whole vocabulary. An absolute median is not comparable + // across vocabulary sizes, since a given token is rarer relative to a larger + // vocabulary; a percentile rank is. + all := slices.Clone(e.Counts) + + slices.Sort(all) + slices.Sort(morphFreq) + slices.Sort(otherFreq) + + row.PctRankMorph = percentileRank(all, median(morphFreq)) + row.PctRankOther = percentileRank(all, median(otherFreq)) + + row.SharedMorph = share(int64(morphShared), int64(row.MorphCount)) + row.SharedOther = share(int64(otherShared), int64(otherCount)) + + return row, nil +} + +// percentileRank is the share of sorted strictly below v, so a higher value +// means a more frequent token. +func percentileRank(sorted []int, v int) float64 { + if len(sorted) == 0 { + return 0 + } + + return 100 * float64(sort.SearchInts(sorted, v)) / float64(len(sorted)) +} + +// TableCRow is the divergence between one tokenizer and a reference. +// +// Everything is measured against the BASELINE segmentation. A pre-token is +// bucketed by how many pieces the baseline produced for it and contributes that +// many token occurrences, so a pre-token that is one token under the baseline +// and two under the variant lands in bucket 1 and counts once. That makes the +// three bucket columns sum to TokensDiff by construction, and keeps the buckets +// the same population as the baseline's table A row. +type TableCRow struct { + Spec TableSpec + Baseline string + + TypesDiff float64 + TokensDiff float64 + + TokensDiffBucket [3]float64 + + // morpheme share of the subword tokens in the differing slice, under each + // side, each normalised by that side's own token count + DiffMorphBase float64 + DiffMorphVar float64 +} + +func BuildTableC(e, base *Encoded, items []mbpe.Chunk) (TableCRow, error) { + if len(e.Sig) != len(base.Sig) { + return TableCRow{}, fmt.Errorf("dictionaries differ in length") + } + + row := TableCRow{Spec: e.Spec, Baseline: base.Spec.Name} + + var typDiff, diffTokens int64 + var diffBucket [3]int64 + var mBase, nBase, mVar, nVar int64 + + for i := range base.Sig { + if e.Sig[i] == base.Sig[i] { + continue + } + + n := int64(items[i].N()) + k := int64(base.Len[i]) + + typDiff++ + diffTokens += n * k + diffBucket[bucket(int(base.Len[i]))] += n * k + + mBase += n * int64(base.MorphN[i]) + nBase += n * k + + mVar += n * int64(e.MorphN[i]) + nVar += n * int64(e.Len[i]) + } + + row.TypesDiff = share(typDiff, int64(len(base.Sig))) + row.TokensDiff = share(diffTokens, base.Tokens) + + for b := range diffBucket { + row.TokensDiffBucket[b] = share(diffBucket[b], base.Tokens) + } + + row.DiffMorphBase = share(mBase, nBase) + row.DiffMorphVar = share(mVar, nVar) + + return row, nil +} + +func share(x, total int64) float64 { + if total == 0 { + return 0 + } + + return 100 * float64(x) / float64(total) +} + +func writeCSV(name string, header []string, rows [][]string) error { + file, err := os.Create(name) + + if err != nil { + return err + } + + defer file.Close() + + w := csv.NewWriter(file) + + if err := w.Write(header); err != nil { + return err + } + + for _, r := range rows { + if err := w.Write(r); err != nil { + return err + } + } + + w.Flush() + + return w.Error() +} + +// percentages are plain numbers, without a percent sign +func f2(v float64) string { return strconv.FormatFloat(v, 'f', 2, 64) } +func f4(v float64) string { return strconv.FormatFloat(v, 'f', 4, 64) } + +func WriteTableA(name string, rows []TableARow) error { + out := make([][]string, 0, len(rows)) + + for _, r := range rows { + out = append(out, []string{ + r.Spec.Name, strconv.Itoa(r.Spec.VocabSize), r.Spec.Alignment, f4(r.Fertility), + f2(r.Types[0]), f2(r.Types[1]), f2(r.Types[2]), + f2(r.Tokens[0]), f2(r.Tokens[1]), f2(r.Tokens[2]), + }) + } + + return writeCSV(name, []string{ + "tokenizer", "vocab_size", "alignment", "fertility", + "types_1", "types_2", "types_3plus", + "tokens_1", "tokens_2", "tokens_3plus", + }, out) +} + +func WriteTableB(name string, rows []TableBRow) error { + out := make([][]string, 0, len(rows)) + + for _, r := range rows { + out = append(out, []string{ + r.Spec.Name, strconv.Itoa(r.Spec.VocabSize), r.Spec.Alignment, f4(r.Fertility), + strconv.Itoa(r.MorphCount), + f2(r.MorphShareUnweighted), f2(r.MorphShareWeighted), + f2(r.PctRankMorph), f2(r.PctRankOther), + f2(r.SharedMorph), f2(r.SharedOther), + }) + } + + return writeCSV(name, []string{ + "tokenizer", "vocab_size", "alignment", "fertility", + "morph_count", "morph_share_unweighted", "morph_share_weighted", + "pct_rank_morph", "pct_rank_other", + "shared_morph", "shared_other", + }, out) +} + +func WriteTableC(name string, rows []TableCRow) error { + out := make([][]string, 0, len(rows)) + + for _, r := range rows { + out = append(out, []string{ + r.Spec.Name, strconv.Itoa(r.Spec.VocabSize), r.Spec.Alignment, r.Baseline, + f2(r.TypesDiff), f2(r.TokensDiff), + f2(r.TokensDiffBucket[0]), f2(r.TokensDiffBucket[1]), f2(r.TokensDiffBucket[2]), + f2(r.DiffMorphBase), f2(r.DiffMorphVar), + }) + } + + return writeCSV(name, []string{ + "tokenizer", "vocab_size", "alignment", "baseline", + "types_diff", "tokens_diff", + "tokens_1_diff", "tokens_2_diff", "tokens_3plus_diff", + "tokens_diff_morph_base", "tokens_diff_morph_var", + }, out) +} |
