1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
|
package entropy
import (
"database/sql"
)
type Tokenizer interface {
Tokenize(string) []int
}
// Criterion turns per-position mean logprobs into per-position boundary
// weights, zero wherever there is no evidence of a boundary.
type Criterion func([]float64) []float64
// ExcessCriterion is Excess bound to a threshold. Weights come out in nats
// below that threshold, so a Segmenter using it wants a cutoff of 0.
func ExcessCriterion(threshold float64) Criterion {
return func(values []float64) []float64 {
return Excess(values, threshold)
}
}
// Segmenter turns surprisal into segments, satisfying the mbpe Segmenter
// interface so it can stand in for Morfessor without touching the trainer.
//
// A boundary is placed wherever the criterion weighs a position above cutoff.
// Note that cutoff applies to the weights, not to the logprobs: with Spikes it
// is a minimum rise in nats and belongs above zero, with ExcessCriterion the
// threshold is already baked in and cutoff belongs at zero.
type Segmenter struct {
tokenizer Tokenizer
db *sql.DB
criterion Criterion
cutoff float64
}
func NewSegmenter(tokenizer Tokenizer, db *sql.DB, criterion Criterion, cutoff float64) *Segmenter {
return &Segmenter{
tokenizer: tokenizer,
db: db,
criterion: criterion,
cutoff: cutoff,
}
}
// Segment reports the segmentation of compound and whether surprisal was known
// for it at all. Compounds the model never saw come back unsegmented and not
// ok, which the trainer already treats as a reason to drop alpha to zero.
//
// The leading whitespace marker the trainer strips is put back before the
// lookup: the logprobs were measured on running text, so the first character
// of a word is only predictable given the space in front of it.
func (s *Segmenter) Segment(compound string) ([]string, bool) {
runes := []rune(compound)
if len(runes) < 2 {
return []string{compound}, false
}
ids := s.tokenizer.Tokenize("Ġ" + compound)
if len(ids) != len(runes)+1 {
return []string{compound}, false // not one token per rune, cannot align
}
values, occurrences, err := MeanLogProbs(ids, s.db)
if err != nil || occurrences == 0 || len(values) != len(ids) {
return []string{compound}, false
}
weights := s.criterion(values)
segments := make([]string, 0, 4)
start := 0
// ids[0] is the whitespace marker and ids[1] the first rune of compound, so
// a boundary there is the start of the chunk and carries no information.
// Skipping it also drops the onset spike, which dominates every sequence.
for i := 2; i < len(ids); i++ {
if weights[i] <= s.cutoff {
continue
}
segments = append(segments, string(runes[start:i-1]))
start = i - 1
}
segments = append(segments, string(runes[start:]))
return segments, true
}
|