summaryrefslogtreecommitdiff
path: root/research/entropy/segmenter.go
diff options
context:
space:
mode:
authorJonas Knobloch <jonas.knobloch@t-online.de>2026-09-09 02:41:24 +0200
committerJonas Knobloch <jonas.knobloch@t-online.de>2026-09-09 02:41:24 +0200
commit3c5f4c4279e6c9d6479e40eff4fbe1add59c1b3a (patch)
treec12ce174c117b0205bc3426b6cc2501f7e709f92 /research/entropy/segmenter.go
parent8fd3f8aeec85f47d43ef36a81456d48640ca33f9 (diff)
WIP
Diffstat (limited to 'research/entropy/segmenter.go')
-rw-r--r--research/entropy/segmenter.go94
1 files changed, 94 insertions, 0 deletions
diff --git a/research/entropy/segmenter.go b/research/entropy/segmenter.go
new file mode 100644
index 0000000..1a15111
--- /dev/null
+++ b/research/entropy/segmenter.go
@@ -0,0 +1,94 @@
+package entropy
+
+import (
+ "database/sql"
+)
+
+type Tokenizer interface {
+ Tokenize(string) []int
+}
+
+// Criterion turns per-position mean logprobs into per-position boundary
+// weights, zero wherever there is no evidence of a boundary.
+type Criterion func([]float64) []float64
+
+// ExcessCriterion is Excess bound to a threshold. Weights come out in nats
+// below that threshold, so a Segmenter using it wants a cutoff of 0.
+func ExcessCriterion(threshold float64) Criterion {
+ return func(values []float64) []float64 {
+ return Excess(values, threshold)
+ }
+}
+
+// Segmenter turns surprisal into segments, satisfying the mbpe Segmenter
+// interface so it can stand in for Morfessor without touching the trainer.
+//
+// A boundary is placed wherever the criterion weighs a position above cutoff.
+// Note that cutoff applies to the weights, not to the logprobs: with Spikes it
+// is a minimum rise in nats and belongs above zero, with ExcessCriterion the
+// threshold is already baked in and cutoff belongs at zero.
+type Segmenter struct {
+ tokenizer Tokenizer
+ db *sql.DB
+ criterion Criterion
+ cutoff float64
+}
+
+func NewSegmenter(tokenizer Tokenizer, db *sql.DB, criterion Criterion, cutoff float64) *Segmenter {
+ return &Segmenter{
+ tokenizer: tokenizer,
+ db: db,
+ criterion: criterion,
+ cutoff: cutoff,
+ }
+}
+
+// Segment reports the segmentation of compound and whether surprisal was known
+// for it at all. Compounds the model never saw come back unsegmented and not
+// ok, which the trainer already treats as a reason to drop alpha to zero.
+//
+// The leading whitespace marker the trainer strips is put back before the
+// lookup: the logprobs were measured on running text, so the first character
+// of a word is only predictable given the space in front of it.
+func (s *Segmenter) Segment(compound string) ([]string, bool) {
+ runes := []rune(compound)
+
+ if len(runes) < 2 {
+ return []string{compound}, false
+ }
+
+ ids := s.tokenizer.Tokenize("Ġ" + compound)
+
+ if len(ids) != len(runes)+1 {
+ return []string{compound}, false // not one token per rune, cannot align
+ }
+
+ values, occurrences, err := MeanLogProbs(ids, s.db)
+
+ if err != nil || occurrences == 0 || len(values) != len(ids) {
+ return []string{compound}, false
+ }
+
+ weights := s.criterion(values)
+
+ segments := make([]string, 0, 4)
+
+ start := 0
+
+ // ids[0] is the whitespace marker and ids[1] the first rune of compound, so
+ // a boundary there is the start of the chunk and carries no information.
+ // Skipping it also drops the onset spike, which dominates every sequence.
+ for i := 2; i < len(ids); i++ {
+ if weights[i] <= s.cutoff {
+ continue
+ }
+
+ segments = append(segments, string(runes[start:i-1]))
+
+ start = i - 1
+ }
+
+ segments = append(segments, string(runes[start:]))
+
+ return segments, true
+}