summaryrefslogtreecommitdiff
path: root/research/entropy/segmenter.go
diff options
context:
space:
mode:
Diffstat (limited to 'research/entropy/segmenter.go')
-rw-r--r--research/entropy/segmenter.go94
1 files changed, 0 insertions, 94 deletions
diff --git a/research/entropy/segmenter.go b/research/entropy/segmenter.go
deleted file mode 100644
index 1a15111..0000000
--- a/research/entropy/segmenter.go
+++ /dev/null
@@ -1,94 +0,0 @@
-package entropy
-
-import (
- "database/sql"
-)
-
-type Tokenizer interface {
- Tokenize(string) []int
-}
-
-// Criterion turns per-position mean logprobs into per-position boundary
-// weights, zero wherever there is no evidence of a boundary.
-type Criterion func([]float64) []float64
-
-// ExcessCriterion is Excess bound to a threshold. Weights come out in nats
-// below that threshold, so a Segmenter using it wants a cutoff of 0.
-func ExcessCriterion(threshold float64) Criterion {
- return func(values []float64) []float64 {
- return Excess(values, threshold)
- }
-}
-
-// Segmenter turns surprisal into segments, satisfying the mbpe Segmenter
-// interface so it can stand in for Morfessor without touching the trainer.
-//
-// A boundary is placed wherever the criterion weighs a position above cutoff.
-// Note that cutoff applies to the weights, not to the logprobs: with Spikes it
-// is a minimum rise in nats and belongs above zero, with ExcessCriterion the
-// threshold is already baked in and cutoff belongs at zero.
-type Segmenter struct {
- tokenizer Tokenizer
- db *sql.DB
- criterion Criterion
- cutoff float64
-}
-
-func NewSegmenter(tokenizer Tokenizer, db *sql.DB, criterion Criterion, cutoff float64) *Segmenter {
- return &Segmenter{
- tokenizer: tokenizer,
- db: db,
- criterion: criterion,
- cutoff: cutoff,
- }
-}
-
-// Segment reports the segmentation of compound and whether surprisal was known
-// for it at all. Compounds the model never saw come back unsegmented and not
-// ok, which the trainer already treats as a reason to drop alpha to zero.
-//
-// The leading whitespace marker the trainer strips is put back before the
-// lookup: the logprobs were measured on running text, so the first character
-// of a word is only predictable given the space in front of it.
-func (s *Segmenter) Segment(compound string) ([]string, bool) {
- runes := []rune(compound)
-
- if len(runes) < 2 {
- return []string{compound}, false
- }
-
- ids := s.tokenizer.Tokenize("Ġ" + compound)
-
- if len(ids) != len(runes)+1 {
- return []string{compound}, false // not one token per rune, cannot align
- }
-
- values, occurrences, err := MeanLogProbs(ids, s.db)
-
- if err != nil || occurrences == 0 || len(values) != len(ids) {
- return []string{compound}, false
- }
-
- weights := s.criterion(values)
-
- segments := make([]string, 0, 4)
-
- start := 0
-
- // ids[0] is the whitespace marker and ids[1] the first rune of compound, so
- // a boundary there is the start of the chunk and carries no information.
- // Skipping it also drops the onset spike, which dominates every sequence.
- for i := 2; i < len(ids); i++ {
- if weights[i] <= s.cutoff {
- continue
- }
-
- segments = append(segments, string(runes[start:i-1]))
-
- start = i - 1
- }
-
- segments = append(segments, string(runes[start:]))
-
- return segments, true
-}