diff options
Diffstat (limited to 'research/entropy/segmenter.go')
| -rw-r--r-- | research/entropy/segmenter.go | 94 |
1 files changed, 94 insertions, 0 deletions
diff --git a/research/entropy/segmenter.go b/research/entropy/segmenter.go new file mode 100644 index 0000000..1a15111 --- /dev/null +++ b/research/entropy/segmenter.go @@ -0,0 +1,94 @@ +package entropy + +import ( + "database/sql" +) + +type Tokenizer interface { + Tokenize(string) []int +} + +// Criterion turns per-position mean logprobs into per-position boundary +// weights, zero wherever there is no evidence of a boundary. +type Criterion func([]float64) []float64 + +// ExcessCriterion is Excess bound to a threshold. Weights come out in nats +// below that threshold, so a Segmenter using it wants a cutoff of 0. +func ExcessCriterion(threshold float64) Criterion { + return func(values []float64) []float64 { + return Excess(values, threshold) + } +} + +// Segmenter turns surprisal into segments, satisfying the mbpe Segmenter +// interface so it can stand in for Morfessor without touching the trainer. +// +// A boundary is placed wherever the criterion weighs a position above cutoff. +// Note that cutoff applies to the weights, not to the logprobs: with Spikes it +// is a minimum rise in nats and belongs above zero, with ExcessCriterion the +// threshold is already baked in and cutoff belongs at zero. +type Segmenter struct { + tokenizer Tokenizer + db *sql.DB + criterion Criterion + cutoff float64 +} + +func NewSegmenter(tokenizer Tokenizer, db *sql.DB, criterion Criterion, cutoff float64) *Segmenter { + return &Segmenter{ + tokenizer: tokenizer, + db: db, + criterion: criterion, + cutoff: cutoff, + } +} + +// Segment reports the segmentation of compound and whether surprisal was known +// for it at all. Compounds the model never saw come back unsegmented and not +// ok, which the trainer already treats as a reason to drop alpha to zero. +// +// The leading whitespace marker the trainer strips is put back before the +// lookup: the logprobs were measured on running text, so the first character +// of a word is only predictable given the space in front of it. +func (s *Segmenter) Segment(compound string) ([]string, bool) { + runes := []rune(compound) + + if len(runes) < 2 { + return []string{compound}, false + } + + ids := s.tokenizer.Tokenize("Ġ" + compound) + + if len(ids) != len(runes)+1 { + return []string{compound}, false // not one token per rune, cannot align + } + + values, occurrences, err := MeanLogProbs(ids, s.db) + + if err != nil || occurrences == 0 || len(values) != len(ids) { + return []string{compound}, false + } + + weights := s.criterion(values) + + segments := make([]string, 0, 4) + + start := 0 + + // ids[0] is the whitespace marker and ids[1] the first rune of compound, so + // a boundary there is the start of the chunk and carries no information. + // Skipping it also drops the onset spike, which dominates every sequence. + for i := 2; i < len(ids); i++ { + if weights[i] <= s.cutoff { + continue + } + + segments = append(segments, string(runes[start:i-1])) + + start = i - 1 + } + + segments = append(segments, string(runes[start:])) + + return segments, true +} |
