summaryrefslogtreecommitdiff
path: root/research/knobloch/tables.go
diff options
context:
space:
mode:
Diffstat (limited to 'research/knobloch/tables.go')
-rw-r--r--research/knobloch/tables.go483
1 files changed, 483 insertions, 0 deletions
diff --git a/research/knobloch/tables.go b/research/knobloch/tables.go
new file mode 100644
index 0000000..c5b2dcf
--- /dev/null
+++ b/research/knobloch/tables.go
@@ -0,0 +1,483 @@
+package knobloch
+
+// Tokenizer characterisation tables. Column meanings are documented in
+// FREQUENCIES.md; keep the two in sync.
+//
+// Terminology, used consistently in identifiers and column names:
+//
+// type a distinct pre-token, i.e. one dictionary entry
+// occurrence a pre-token weighted by its corpus count
+// token a subword unit, i.e. a vocabulary entry
+//
+// Morpheme labels attach to subword tokens, never to pre-token types, so the
+// morpheme columns of table B are named morph_* rather than types_*.
+//
+// Tokenizers are compared by token string, never by id: the same string carries
+// a different id in every vocabulary, so comparing ids would measure id
+// reassignment rather than segmentation.
+
+import (
+ "encoding/csv"
+ "encoding/json"
+ "fmt"
+ "hash/fnv"
+ "os"
+ "slices"
+ "sort"
+ "strconv"
+
+ "github.com/jonasknobloch/mbpe"
+ "go.jknobloc.com/x/shelf"
+ "go.jknobloc.com/x/tokenizer/bpe"
+)
+
+// TableSpec identifies one tokenizer to profile.
+type TableSpec struct {
+ Name string // tokenizer label, e.g. m050
+ Alignment string // alpha as a plain number; direction is carried by Inverted
+ Inverted bool
+ VocabSize int
+ Dir shelf.Item // directory holding vocab.json and merges.txt
+}
+
+// Encoded is one tokenizer's view of the dictionary.
+//
+// Comparisons need to know, per dictionary entry, whether two tokenizers agree
+// and how many pieces each produced. Retaining the pieces themselves costs
+// hundreds of megabytes per tokenizer, which does not fit for a whole
+// vocabulary-size family at once, so each entry is reduced to a hash of its
+// segmentation plus two small counts. That is all tables A, B and C read.
+type Encoded struct {
+ Spec TableSpec
+
+ // corpus count per vocab id, and the id-to-string map
+ Counts []int
+ Itoa map[int64]string
+
+ Words int64
+ Tokens int64
+
+ // pieces-per-type distribution, bucketed 1 / 2 / 3+
+ TypeBucket [3]int64
+ TokenBucket [3]int64
+
+ // per dictionary entry, in dict order
+ Sig []uint64 // hash of the piece strings
+ Len []uint16 // number of pieces
+ MorphN []uint16 // number of pieces labelled morpheme
+}
+
+// Fertility is subword tokens per pre-token occurrence.
+func (e *Encoded) Fertility() float64 {
+ if e.Words == 0 {
+ return 0
+ }
+
+ return float64(e.Tokens) / float64(e.Words)
+}
+
+// EncodeDict segments every dictionary entry with one tokenizer.
+func EncodeDict(spec TableSpec, items []mbpe.Chunk) (*Encoded, error) {
+ morph, err := morphemes()
+
+ if err != nil {
+ return nil, err
+ }
+
+ tok, err := bpe.NewTokenizerFromFiles(
+ shelf.Abs(spec.Dir+"/vocab.json"), shelf.Abs(spec.Dir+"/merges.txt"),
+ bpe.Config{Recover: false})
+
+ if err != nil {
+ return nil, err
+ }
+
+ // the dictionary is already pre-tokenised
+ bpe.MBPE(tok).SetPreTokenizer(&NoPreTok{})
+
+ e := &Encoded{
+ Spec: spec,
+ Counts: make([]int, len(bpe.Vocab(tok))),
+ Itoa: bpe.Itoa(tok),
+ Sig: make([]uint64, len(items)),
+ Len: make([]uint16, len(items)),
+ MorphN: make([]uint16, len(items)),
+ }
+
+ h := fnv.New64a()
+
+ for i, v := range items {
+ ids := tok.Encode(v.Src())
+
+ h.Reset()
+
+ var morphN uint16
+
+ for _, id := range ids {
+ s := e.Itoa[int64(id)]
+
+ e.Counts[id] += v.N()
+
+ // the separator keeps ab|c distinct from a|bc
+ h.Write([]byte(s))
+ h.Write([]byte{0})
+
+ if _, ok := morph[s]; ok {
+ morphN++
+ }
+ }
+
+ e.Sig[i] = h.Sum64()
+ e.Len[i] = uint16(len(ids))
+ e.MorphN[i] = morphN
+
+ n := int64(v.N())
+ k := int64(len(ids))
+ b := bucket(len(ids))
+
+ e.Words += n
+ e.Tokens += n * k
+
+ e.TypeBucket[b]++
+ e.TokenBucket[b] += n * k
+ }
+
+ return e, nil
+}
+
+// bucket maps a piece count onto the 1 / 2 / 3+ buckets used by every table.
+func bucket(k int) int {
+ switch {
+ case k == 1:
+ return 0
+ case k == 2:
+ return 1
+ default:
+ return 2
+ }
+}
+
+// VocabIntersection returns the token strings present in every one of the given
+// vocabularies. Reads vocab.json only, so it never touches the corpus.
+func VocabIntersection(dirs []shelf.Item) (map[string]struct{}, error) {
+ if len(dirs) == 0 {
+ return nil, fmt.Errorf("no vocabularies given")
+ }
+
+ var keep map[string]struct{}
+
+ for _, dir := range dirs {
+ file, err := os.Open(shelf.Abs(dir + "/vocab.json"))
+
+ if err != nil {
+ return nil, err
+ }
+
+ var v map[string]int64
+
+ err = json.NewDecoder(file).Decode(&v)
+
+ file.Close()
+
+ if err != nil {
+ return nil, err
+ }
+
+ if keep == nil {
+ keep = make(map[string]struct{}, len(v))
+
+ for token := range v {
+ keep[token] = struct{}{}
+ }
+
+ continue
+ }
+
+ for token := range keep {
+ if _, ok := v[token]; !ok {
+ delete(keep, token)
+ }
+ }
+ }
+
+ return keep, nil
+}
+
+// TableARow is the segmentation profile of one tokenizer.
+type TableARow struct {
+ Spec TableSpec
+ Fertility float64
+
+ Types [3]float64 // share of pre-token types
+ Tokens [3]float64 // share of subword token occurrences
+}
+
+func BuildTableA(e *Encoded) TableARow {
+ row := TableARow{Spec: e.Spec, Fertility: e.Fertility()}
+
+ var typeTotal, tokenTotal int64
+
+ for b := range e.TypeBucket {
+ typeTotal += e.TypeBucket[b]
+ tokenTotal += e.TokenBucket[b]
+ }
+
+ for b := range e.TypeBucket {
+ row.Types[b] = share(e.TypeBucket[b], typeTotal)
+ row.Tokens[b] = share(e.TokenBucket[b], tokenTotal)
+ }
+
+ return row
+}
+
+// TableBRow describes what a vocabulary is made of.
+type TableBRow struct {
+ Spec TableSpec
+ Fertility float64
+
+ MorphCount int
+ MorphShareUnweighted float64
+ MorphShareWeighted float64
+
+ PctRankMorph float64
+ PctRankOther float64
+
+ SharedMorph float64
+ SharedOther float64
+}
+
+// BuildTableB profiles one vocabulary. shared is the intersection of every
+// vocabulary at this size, which is what the two shared_* columns measure
+// against — so the baseline is just another tokenizer and gets a real value.
+func BuildTableB(e *Encoded, shared map[string]struct{}) (TableBRow, error) {
+ morph, err := morphemes()
+
+ if err != nil {
+ return TableBRow{}, err
+ }
+
+ row := TableBRow{Spec: e.Spec, Fertility: e.Fertility()}
+
+ var morphTokens, otherTokens int64
+ var morphFreq, otherFreq []int
+ var morphShared, otherShared, otherCount int
+
+ for id, f := range e.Counts {
+ token := e.Itoa[int64(id)]
+
+ _, isMorph := morph[token]
+ _, inShared := shared[token]
+
+ if isMorph {
+ row.MorphCount++
+ morphTokens += int64(f)
+ morphFreq = append(morphFreq, f)
+
+ if inShared {
+ morphShared++
+ }
+ } else {
+ otherCount++
+ otherTokens += int64(f)
+ otherFreq = append(otherFreq, f)
+
+ if inShared {
+ otherShared++
+ }
+ }
+ }
+
+ row.MorphShareUnweighted = share(int64(row.MorphCount), int64(len(e.Counts)))
+ row.MorphShareWeighted = share(morphTokens, morphTokens+otherTokens)
+
+ // Percentile rank of each group's median token within the corpus frequency
+ // distribution of the whole vocabulary. An absolute median is not comparable
+ // across vocabulary sizes, since a given token is rarer relative to a larger
+ // vocabulary; a percentile rank is.
+ all := slices.Clone(e.Counts)
+
+ slices.Sort(all)
+ slices.Sort(morphFreq)
+ slices.Sort(otherFreq)
+
+ row.PctRankMorph = percentileRank(all, median(morphFreq))
+ row.PctRankOther = percentileRank(all, median(otherFreq))
+
+ row.SharedMorph = share(int64(morphShared), int64(row.MorphCount))
+ row.SharedOther = share(int64(otherShared), int64(otherCount))
+
+ return row, nil
+}
+
+// percentileRank is the share of sorted strictly below v, so a higher value
+// means a more frequent token.
+func percentileRank(sorted []int, v int) float64 {
+ if len(sorted) == 0 {
+ return 0
+ }
+
+ return 100 * float64(sort.SearchInts(sorted, v)) / float64(len(sorted))
+}
+
+// TableCRow is the divergence between one tokenizer and a reference.
+//
+// Everything is measured against the BASELINE segmentation. A pre-token is
+// bucketed by how many pieces the baseline produced for it and contributes that
+// many token occurrences, so a pre-token that is one token under the baseline
+// and two under the variant lands in bucket 1 and counts once. That makes the
+// three bucket columns sum to TokensDiff by construction, and keeps the buckets
+// the same population as the baseline's table A row.
+type TableCRow struct {
+ Spec TableSpec
+ Baseline string
+
+ TypesDiff float64
+ TokensDiff float64
+
+ TokensDiffBucket [3]float64
+
+ // morpheme share of the subword tokens in the differing slice, under each
+ // side, each normalised by that side's own token count
+ DiffMorphBase float64
+ DiffMorphVar float64
+}
+
+func BuildTableC(e, base *Encoded, items []mbpe.Chunk) (TableCRow, error) {
+ if len(e.Sig) != len(base.Sig) {
+ return TableCRow{}, fmt.Errorf("dictionaries differ in length")
+ }
+
+ row := TableCRow{Spec: e.Spec, Baseline: base.Spec.Name}
+
+ var typDiff, diffTokens int64
+ var diffBucket [3]int64
+ var mBase, nBase, mVar, nVar int64
+
+ for i := range base.Sig {
+ if e.Sig[i] == base.Sig[i] {
+ continue
+ }
+
+ n := int64(items[i].N())
+ k := int64(base.Len[i])
+
+ typDiff++
+ diffTokens += n * k
+ diffBucket[bucket(int(base.Len[i]))] += n * k
+
+ mBase += n * int64(base.MorphN[i])
+ nBase += n * k
+
+ mVar += n * int64(e.MorphN[i])
+ nVar += n * int64(e.Len[i])
+ }
+
+ row.TypesDiff = share(typDiff, int64(len(base.Sig)))
+ row.TokensDiff = share(diffTokens, base.Tokens)
+
+ for b := range diffBucket {
+ row.TokensDiffBucket[b] = share(diffBucket[b], base.Tokens)
+ }
+
+ row.DiffMorphBase = share(mBase, nBase)
+ row.DiffMorphVar = share(mVar, nVar)
+
+ return row, nil
+}
+
+func share(x, total int64) float64 {
+ if total == 0 {
+ return 0
+ }
+
+ return 100 * float64(x) / float64(total)
+}
+
+func writeCSV(name string, header []string, rows [][]string) error {
+ file, err := os.Create(name)
+
+ if err != nil {
+ return err
+ }
+
+ defer file.Close()
+
+ w := csv.NewWriter(file)
+
+ if err := w.Write(header); err != nil {
+ return err
+ }
+
+ for _, r := range rows {
+ if err := w.Write(r); err != nil {
+ return err
+ }
+ }
+
+ w.Flush()
+
+ return w.Error()
+}
+
+// percentages are plain numbers, without a percent sign
+func f2(v float64) string { return strconv.FormatFloat(v, 'f', 2, 64) }
+func f4(v float64) string { return strconv.FormatFloat(v, 'f', 4, 64) }
+
+func WriteTableA(name string, rows []TableARow) error {
+ out := make([][]string, 0, len(rows))
+
+ for _, r := range rows {
+ out = append(out, []string{
+ r.Spec.Name, strconv.Itoa(r.Spec.VocabSize), r.Spec.Alignment, f4(r.Fertility),
+ f2(r.Types[0]), f2(r.Types[1]), f2(r.Types[2]),
+ f2(r.Tokens[0]), f2(r.Tokens[1]), f2(r.Tokens[2]),
+ })
+ }
+
+ return writeCSV(name, []string{
+ "tokenizer", "vocab_size", "alignment", "fertility",
+ "types_1", "types_2", "types_3plus",
+ "tokens_1", "tokens_2", "tokens_3plus",
+ }, out)
+}
+
+func WriteTableB(name string, rows []TableBRow) error {
+ out := make([][]string, 0, len(rows))
+
+ for _, r := range rows {
+ out = append(out, []string{
+ r.Spec.Name, strconv.Itoa(r.Spec.VocabSize), r.Spec.Alignment, f4(r.Fertility),
+ strconv.Itoa(r.MorphCount),
+ f2(r.MorphShareUnweighted), f2(r.MorphShareWeighted),
+ f2(r.PctRankMorph), f2(r.PctRankOther),
+ f2(r.SharedMorph), f2(r.SharedOther),
+ })
+ }
+
+ return writeCSV(name, []string{
+ "tokenizer", "vocab_size", "alignment", "fertility",
+ "morph_count", "morph_share_unweighted", "morph_share_weighted",
+ "pct_rank_morph", "pct_rank_other",
+ "shared_morph", "shared_other",
+ }, out)
+}
+
+func WriteTableC(name string, rows []TableCRow) error {
+ out := make([][]string, 0, len(rows))
+
+ for _, r := range rows {
+ out = append(out, []string{
+ r.Spec.Name, strconv.Itoa(r.Spec.VocabSize), r.Spec.Alignment, r.Baseline,
+ f2(r.TypesDiff), f2(r.TokensDiff),
+ f2(r.TokensDiffBucket[0]), f2(r.TokensDiffBucket[1]), f2(r.TokensDiffBucket[2]),
+ f2(r.DiffMorphBase), f2(r.DiffMorphVar),
+ })
+ }
+
+ return writeCSV(name, []string{
+ "tokenizer", "vocab_size", "alignment", "baseline",
+ "types_diff", "tokens_diff",
+ "tokens_1_diff", "tokens_2_diff", "tokens_3plus_diff",
+ "tokens_diff_morph_base", "tokens_diff_morph_var",
+ }, out)
+}