package entropy import ( "database/sql" ) type Tokenizer interface { Tokenize(string) []int } // Criterion turns per-position mean logprobs into per-position boundary // weights, zero wherever there is no evidence of a boundary. type Criterion func([]float64) []float64 // ExcessCriterion is Excess bound to a threshold. Weights come out in nats // below that threshold, so a Segmenter using it wants a cutoff of 0. func ExcessCriterion(threshold float64) Criterion { return func(values []float64) []float64 { return Excess(values, threshold) } } // Segmenter turns surprisal into segments, satisfying the mbpe Segmenter // interface so it can stand in for Morfessor without touching the trainer. // // A boundary is placed wherever the criterion weighs a position above cutoff. // Note that cutoff applies to the weights, not to the logprobs: with Spikes it // is a minimum rise in nats and belongs above zero, with ExcessCriterion the // threshold is already baked in and cutoff belongs at zero. type Segmenter struct { tokenizer Tokenizer db *sql.DB criterion Criterion cutoff float64 } func NewSegmenter(tokenizer Tokenizer, db *sql.DB, criterion Criterion, cutoff float64) *Segmenter { return &Segmenter{ tokenizer: tokenizer, db: db, criterion: criterion, cutoff: cutoff, } } // Segment reports the segmentation of compound and whether surprisal was known // for it at all. Compounds the model never saw come back unsegmented and not // ok, which the trainer already treats as a reason to drop alpha to zero. // // The leading whitespace marker the trainer strips is put back before the // lookup: the logprobs were measured on running text, so the first character // of a word is only predictable given the space in front of it. func (s *Segmenter) Segment(compound string) ([]string, bool) { runes := []rune(compound) if len(runes) < 2 { return []string{compound}, false } ids := s.tokenizer.Tokenize("Ġ" + compound) if len(ids) != len(runes)+1 { return []string{compound}, false // not one token per rune, cannot align } values, occurrences, err := MeanLogProbs(ids, s.db) if err != nil || occurrences == 0 || len(values) != len(ids) { return []string{compound}, false } weights := s.criterion(values) segments := make([]string, 0, 4) start := 0 // ids[0] is the whitespace marker and ids[1] the first rune of compound, so // a boundary there is the start of the chunk and carries no information. // Skipping it also drops the onset spike, which dominates every sequence. for i := 2; i < len(ids); i++ { if weights[i] <= s.cutoff { continue } segments = append(segments, string(runes[start:i-1])) start = i - 1 } segments = append(segments, string(runes[start:])) return segments, true }