diff options
Diffstat (limited to 'research/entropy/segmenter.go')
| -rw-r--r-- | research/entropy/segmenter.go | 94 |
1 files changed, 0 insertions, 94 deletions
diff --git a/research/entropy/segmenter.go b/research/entropy/segmenter.go deleted file mode 100644 index 1a15111..0000000 --- a/research/entropy/segmenter.go +++ /dev/null @@ -1,94 +0,0 @@ -package entropy - -import ( - "database/sql" -) - -type Tokenizer interface { - Tokenize(string) []int -} - -// Criterion turns per-position mean logprobs into per-position boundary -// weights, zero wherever there is no evidence of a boundary. -type Criterion func([]float64) []float64 - -// ExcessCriterion is Excess bound to a threshold. Weights come out in nats -// below that threshold, so a Segmenter using it wants a cutoff of 0. -func ExcessCriterion(threshold float64) Criterion { - return func(values []float64) []float64 { - return Excess(values, threshold) - } -} - -// Segmenter turns surprisal into segments, satisfying the mbpe Segmenter -// interface so it can stand in for Morfessor without touching the trainer. -// -// A boundary is placed wherever the criterion weighs a position above cutoff. -// Note that cutoff applies to the weights, not to the logprobs: with Spikes it -// is a minimum rise in nats and belongs above zero, with ExcessCriterion the -// threshold is already baked in and cutoff belongs at zero. -type Segmenter struct { - tokenizer Tokenizer - db *sql.DB - criterion Criterion - cutoff float64 -} - -func NewSegmenter(tokenizer Tokenizer, db *sql.DB, criterion Criterion, cutoff float64) *Segmenter { - return &Segmenter{ - tokenizer: tokenizer, - db: db, - criterion: criterion, - cutoff: cutoff, - } -} - -// Segment reports the segmentation of compound and whether surprisal was known -// for it at all. Compounds the model never saw come back unsegmented and not -// ok, which the trainer already treats as a reason to drop alpha to zero. -// -// The leading whitespace marker the trainer strips is put back before the -// lookup: the logprobs were measured on running text, so the first character -// of a word is only predictable given the space in front of it. -func (s *Segmenter) Segment(compound string) ([]string, bool) { - runes := []rune(compound) - - if len(runes) < 2 { - return []string{compound}, false - } - - ids := s.tokenizer.Tokenize("Ġ" + compound) - - if len(ids) != len(runes)+1 { - return []string{compound}, false // not one token per rune, cannot align - } - - values, occurrences, err := MeanLogProbs(ids, s.db) - - if err != nil || occurrences == 0 || len(values) != len(ids) { - return []string{compound}, false - } - - weights := s.criterion(values) - - segments := make([]string, 0, 4) - - start := 0 - - // ids[0] is the whitespace marker and ids[1] the first rune of compound, so - // a boundary there is the start of the chunk and carries no information. - // Skipping it also drops the onset spike, which dominates every sequence. - for i := 2; i < len(ids); i++ { - if weights[i] <= s.cutoff { - continue - } - - segments = append(segments, string(runes[start:i-1])) - - start = i - 1 - } - - segments = append(segments, string(runes[start:])) - - return segments, true -} |
