mirror of
https://github.com/junegunn/fzf
synced 2026-08-08 19:41:43 +00:00
Prefilter rune-mode input
asciiFuzzyIndex gave up on non-ASCII lines, so every item ran the full score matrix. A []rune is a fixed 4-byte stride, so the SIMD byte scanners can run over it directly: find the low byte, then confirm 4-byte alignment and three zero bytes. Case folding and normalization can turn a non-ASCII rune into the ASCII char being searched, which the scan cannot see. ToChars now flags lines holding such a rune and those keep the old path. Normalization is Latin-only, so Hangul, CJK, Cyrillic, Greek, Hebrew, Arabic, Thai, kana and emoji never set the flag. Chars had no spare padding, so inBytes moves into a flags byte. Measured on 1.4M-line corpora: - Mostly-ASCII paths behind a Hangul prefix: 'conf' 1.8x, 'binutils' 2.9x, 'ltversion' 4.0x, no-match 8.4x - Every line CJK: 17x on both matching and non-matching queries - ASCII input unchanged, non-ASCII patterns not covered yet
This commit is contained in:
@@ -102,7 +102,7 @@ itest:
|
||||
# FUZZTIME (e.g. make fuzz FUZZTIME=5m).
|
||||
FUZZTIME ?= 30s
|
||||
fuzz:
|
||||
@for t in FuzzFuzzyMatchV2Single FuzzFuzzyMatchV2Two; do \
|
||||
@for t in FuzzFuzzyMatchV2Single FuzzFuzzyMatchV2Two FuzzRunePrefilter; do \
|
||||
echo "== $$t =="; \
|
||||
$(GO) test -run '^$$' -fuzz "^$$t$$" -fuzztime $(FUZZTIME) ./src/algo || exit 1; \
|
||||
done
|
||||
|
||||
+46
-7
@@ -345,15 +345,53 @@ func isAscii(runes []rune) bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// runeFuzzyIndex is asciiFuzzyIndex for rune-mode input. Only valid when the
|
||||
// input cannot fold to ASCII, see the caller.
|
||||
func runeFuzzyIndex(input *util.Chars, pattern []rune, caseSensitive bool) (int, int) {
|
||||
runes := input.Runes()
|
||||
firstIdx, idx, lastIdx := 0, 0, 0
|
||||
var b byte
|
||||
for pidx := range pattern {
|
||||
b = byte(pattern[pidx])
|
||||
idx = indexAsciiRune(runes, caseSensitive, b, idx)
|
||||
if idx < 0 {
|
||||
return -1, -1
|
||||
}
|
||||
if pidx == 0 && idx > 0 {
|
||||
// Step back to find the right bonus point
|
||||
firstIdx = idx - 1
|
||||
}
|
||||
lastIdx = idx
|
||||
idx++
|
||||
}
|
||||
|
||||
// Find the last appearance of the last character of the pattern to limit
|
||||
// the search scope
|
||||
if lastIdx+1 < len(runes) {
|
||||
if end := lastIndexAsciiRune(runes, caseSensitive, b, lastIdx+1); end >= 0 {
|
||||
return firstIdx, end + 1
|
||||
}
|
||||
}
|
||||
return firstIdx, lastIdx + 1
|
||||
}
|
||||
|
||||
func asciiFuzzyIndex(input *util.Chars, pattern []rune, caseSensitive bool) (int, int) {
|
||||
// Can't determine
|
||||
if !input.IsBytes() {
|
||||
if !isAscii(pattern) {
|
||||
// An ASCII string cannot contain a non-ASCII character
|
||||
if input.IsBytes() {
|
||||
return -1, -1
|
||||
}
|
||||
// Rune input with a non-ASCII pattern is not filtered yet
|
||||
return 0, input.Length()
|
||||
}
|
||||
|
||||
// Not possible
|
||||
if !isAscii(pattern) {
|
||||
return -1, -1
|
||||
if !input.IsBytes() {
|
||||
// Case folding or normalization can turn a non-ASCII rune into the
|
||||
// ASCII character we are looking for, which the scan cannot see
|
||||
if disableRunePrefilter || input.MayFoldToAscii() {
|
||||
return 0, input.Length()
|
||||
}
|
||||
return runeFuzzyIndex(input, pattern, caseSensitive)
|
||||
}
|
||||
|
||||
firstIdx, idx, lastIdx := 0, 0, 0
|
||||
@@ -466,8 +504,9 @@ func fuzzyMatchV2Single(caseSensitive bool, forward bool, input *util.Chars, b b
|
||||
// Test hooks: force the general path instead of a fast path, so the two can
|
||||
// be compared for equivalence.
|
||||
var (
|
||||
disableSingle bool
|
||||
disableTwo bool
|
||||
disableSingle bool
|
||||
disableTwo bool
|
||||
disableRunePrefilter bool
|
||||
)
|
||||
|
||||
// fuzzyMatchV2Two is a fused fast path for a two-character ASCII pattern on
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
//go:build !386 && !amd64 && !arm64
|
||||
|
||||
package algo
|
||||
|
||||
// The byte-view scanners in runeindex_x86.go reinterpret a []rune as
|
||||
// little-endian 4-byte lanes, which is not valid everywhere. Elsewhere the
|
||||
// reference scanners are the implementation.
|
||||
|
||||
func indexAsciiRune(runes []rune, caseSensitive bool, b byte, from int) int {
|
||||
return indexAsciiRuneRef(runes, caseSensitive, b, from)
|
||||
}
|
||||
|
||||
func lastIndexAsciiRune(runes []rune, caseSensitive bool, b byte, from int) int {
|
||||
return lastIndexAsciiRuneRef(runes, caseSensitive, b, from)
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
package algo
|
||||
|
||||
// Reference scanners over a []rune, with no representation tricks.
|
||||
//
|
||||
// They have two roles. Where reinterpreting a []rune as little-endian bytes is
|
||||
// not valid, they are the shipped implementation, via runeindex_others.go.
|
||||
// Everywhere else, the tests feed the same inputs to these and to the byte-view
|
||||
// scanners in runeindex_x86.go and require identical answers.
|
||||
//
|
||||
// They carry no build tag so that both roles hold on every platform. Otherwise
|
||||
// the portable build would be code that nothing here ever runs.
|
||||
|
||||
func indexAsciiRuneRef(runes []rune, caseSensitive bool, b byte, from int) int {
|
||||
lower, upper := rune(b), rune(-1)
|
||||
if !caseSensitive && b >= 'a' && b <= 'z' {
|
||||
upper = rune(b - 32)
|
||||
}
|
||||
for i := from; i < len(runes); i++ {
|
||||
if runes[i] == lower || runes[i] == upper {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
func lastIndexAsciiRuneRef(runes []rune, caseSensitive bool, b byte, from int) int {
|
||||
lower, upper := rune(b), rune(-1)
|
||||
if !caseSensitive && b >= 'a' && b <= 'z' {
|
||||
upper = rune(b - 32)
|
||||
}
|
||||
for i := len(runes) - 1; i >= from; i-- {
|
||||
if runes[i] == lower || runes[i] == upper {
|
||||
return i
|
||||
}
|
||||
}
|
||||
return -1
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
//go:build 386 || amd64 || arm64
|
||||
|
||||
package algo
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
// On these architectures a []rune is a little-endian array of 4-byte lanes, so
|
||||
// an ASCII rune is the byte itself followed by three zero bytes at a 4-byte
|
||||
// aligned offset. That lets the SIMD byte scanners run over the rune array
|
||||
// directly: find the low byte, then confirm alignment and the three zeroes.
|
||||
// A byte equal to the needle can also appear as the low byte of a multi-byte
|
||||
// rune (0x0165 has low byte 'e'), which those two checks reject.
|
||||
|
||||
func runeBytes(runes []rune) []byte {
|
||||
return unsafe.Slice((*byte)(unsafe.Pointer(unsafe.SliceData(runes))), len(runes)*4)
|
||||
}
|
||||
|
||||
// indexAsciiRune returns the index of the first rune equal to b, or to its
|
||||
// uppercase form when ignoring case, at or after rune index from.
|
||||
func indexAsciiRune(runes []rune, caseSensitive bool, b byte, from int) int {
|
||||
view := runeBytes(runes)
|
||||
both := !caseSensitive && b >= 'a' && b <= 'z'
|
||||
for off := from * 4; off < len(view); {
|
||||
var idx int
|
||||
if both {
|
||||
idx = IndexByteTwo(view[off:], b, b-32)
|
||||
} else {
|
||||
idx = bytes.IndexByte(view[off:], b)
|
||||
}
|
||||
if idx < 0 {
|
||||
return -1
|
||||
}
|
||||
pos := off + idx
|
||||
if pos&3 == 0 && view[pos+1]|view[pos+2]|view[pos+3] == 0 {
|
||||
return pos >> 2
|
||||
}
|
||||
off = pos + 1
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
// lastIndexAsciiRune is indexAsciiRune scanning backwards from the end.
|
||||
func lastIndexAsciiRune(runes []rune, caseSensitive bool, b byte, from int) int {
|
||||
view := runeBytes(runes)[from*4:]
|
||||
both := !caseSensitive && b >= 'a' && b <= 'z'
|
||||
for end := len(view); end > 0; {
|
||||
var idx int
|
||||
if both {
|
||||
idx = lastIndexByteTwo(view[:end], b, b-32)
|
||||
} else {
|
||||
idx = bytes.LastIndexByte(view[:end], b)
|
||||
}
|
||||
if idx < 0 {
|
||||
return -1
|
||||
}
|
||||
if idx&3 == 0 && view[idx+1]|view[idx+2]|view[idx+3] == 0 {
|
||||
return from + idx>>2
|
||||
}
|
||||
end = idx
|
||||
}
|
||||
return -1
|
||||
}
|
||||
@@ -0,0 +1,325 @@
|
||||
package algo
|
||||
|
||||
// Correctness tests for the rune-array prefilter (Step C).
|
||||
//
|
||||
// The prefilter may narrow the search scope but must never change a Result or
|
||||
// its positions, and must never reject an item the general path would match.
|
||||
// Each result is compared against the same code with the prefilter disabled.
|
||||
|
||||
import (
|
||||
"math/rand"
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode"
|
||||
"unicode/utf8"
|
||||
|
||||
"github.com/junegunn/fzf/src/util"
|
||||
)
|
||||
|
||||
// foldForTest mirrors what Phase 2 does to a non-ASCII text rune: lowercase if
|
||||
// uppercase, then normalize.
|
||||
func foldForTest(r rune, normalize bool) rune {
|
||||
if charClassOfNonAscii(r) == charUpper {
|
||||
r = unicode.To(unicode.LowerCase, r)
|
||||
}
|
||||
if normalize {
|
||||
r = normalizeRune(r)
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
// The prefilter is only safe on items whose runes cannot become ASCII. This
|
||||
// pins util.MayFoldToAscii as a superset of the runes that actually can, over
|
||||
// the whole Unicode range and both normalization modes. If normalize.go or the
|
||||
// Go unicode tables change, this fails.
|
||||
func TestMayFoldToAsciiIsSuperset(t *testing.T) {
|
||||
missed := 0
|
||||
for r := rune(utf8.RuneSelf); r <= unicode.MaxRune; r++ {
|
||||
if r >= 0xD800 && r <= 0xDFFF {
|
||||
continue
|
||||
}
|
||||
for _, normalize := range []bool{true, false} {
|
||||
if foldForTest(r, normalize) < utf8.RuneSelf && !util.MayFoldToAscii(r) {
|
||||
if missed++; missed < 10 {
|
||||
t.Errorf("U+%04X folds to ASCII (normalize=%v) but MayFoldToAscii is false", r, normalize)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if missed > 0 {
|
||||
t.Fatalf("%d runes fold to ASCII without being flagged", missed)
|
||||
}
|
||||
}
|
||||
|
||||
// Scripts that must stay unflagged, otherwise the prefilter never engages for
|
||||
// them and Step C buys nothing.
|
||||
func TestMayFoldToAsciiExcludesMajorScripts(t *testing.T) {
|
||||
for _, s := range []struct {
|
||||
name string
|
||||
lo, hi rune
|
||||
}{
|
||||
{"Cyrillic", 0x0400, 0x04FF}, {"Greek", 0x0370, 0x03FF}, {"Hebrew", 0x0590, 0x05FF},
|
||||
{"Arabic", 0x0600, 0x06FF}, {"Thai", 0x0E00, 0x0E7F}, {"Devanagari", 0x0900, 0x097F},
|
||||
{"CJK", 0x4E00, 0x9FFF}, {"Hangul", 0xAC00, 0xD7A3}, {"kana", 0x3040, 0x30FF},
|
||||
{"box drawing", 0x2500, 0x257F}, {"emoji", 0x1F300, 0x1FAFF},
|
||||
// These sit between the Latin blocks and were swallowed by an earlier,
|
||||
// wider grouping of foldableRanges. General Punctuation is the costly
|
||||
// one: curly quotes, en and em dashes and the ellipsis live there.
|
||||
{"Greek Extended", 0x1F00, 0x1FFF}, {"General Punctuation", 0x2000, 0x206F},
|
||||
{"Currency Symbols", 0x20A0, 0x20CF}, {"CJK Symbols", 0x3000, 0x303F},
|
||||
} {
|
||||
for r := s.lo; r <= s.hi; r++ {
|
||||
if util.MayFoldToAscii(r) {
|
||||
t.Errorf("%s U+%04X should not be flagged foldable", s.name, r)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The byte-view scan must agree with the shipped reference scanners. The interesting inputs
|
||||
// are runes whose low byte collides with the needle (U+0165 has low byte 'e')
|
||||
// and runes sharing a lane offset, which the alignment and zero checks reject.
|
||||
func TestIndexAsciiRuneMatchesReference(t *testing.T) {
|
||||
rng := rand.New(rand.NewSource(3))
|
||||
alphabet := []rune{'a', 'A', 'e', 'E', '/', '1', 0x0165, 0x00E9, 0x4E00, 0xD55C,
|
||||
0x1F389, 0x0065 + 0x100, 0x0041 + 0x100, 0x2F65}
|
||||
for trial := range 20000 {
|
||||
n := rng.Intn(24)
|
||||
runes := make([]rune, n)
|
||||
for i := range runes {
|
||||
runes[i] = alphabet[rng.Intn(len(alphabet))]
|
||||
}
|
||||
b := []byte{'a', 'e', 'A', 'E', '/', '1'}[rng.Intn(6)]
|
||||
cs := rng.Intn(2) == 0
|
||||
from := 0
|
||||
if n > 0 {
|
||||
from = rng.Intn(n)
|
||||
}
|
||||
if got, exp := indexAsciiRune(runes, cs, b, from), indexAsciiRuneRef(runes, cs, b, from); got != exp {
|
||||
t.Fatalf("trial %d: indexAsciiRune(%U, cs=%v, %q, %d) = %d, expected %d", trial, runes, cs, b, from, got, exp)
|
||||
}
|
||||
if got, exp := lastIndexAsciiRune(runes, cs, b, from), lastIndexAsciiRuneRef(runes, cs, b, from); got != exp {
|
||||
t.Fatalf("trial %d: lastIndexAsciiRune(%U, cs=%v, %q, %d) = %d, expected %d", trial, runes, cs, b, from, got, exp)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Differential test: the prefilter must not change any Result or position.
|
||||
// Corpora deliberately mix scripts that clear the foldable bit (CJK, Hangul,
|
||||
// Cyrillic, emoji) with scripts that set it (accented Latin, fullwidth), so
|
||||
// both the engaged and the bypassed path are exercised.
|
||||
func TestRunePrefilterEquivalence(t *testing.T) {
|
||||
t.Cleanup(func() { disableRunePrefilter = false })
|
||||
rng := rand.New(rand.NewSource(4))
|
||||
parts := []string{
|
||||
"src", "util", "conf", "a", "e", "E", "A", "/", "_", "1", " ",
|
||||
"漢字", "한글", "мир", "ελλ", "🎉", "café", "Müller", "naïve", "full",
|
||||
"Å", "İ", "K", "ǰ", "ff",
|
||||
}
|
||||
patterns := []string{"a", "e", "conf", "src/util", "ae", "A", "E", "K", "k", "i", "//", "zz", "s l",
|
||||
// non-ASCII patterns, the Step G path
|
||||
"漢", "漢字", "한", "한글", "мир", "м", "ελλ", "🎉", "é", "ß", "Å", "İ", "f",
|
||||
"漢a", "a漢", "한글/src", "🎉e"}
|
||||
|
||||
slab := util.MakeSlab(100*1024, 2048)
|
||||
engaged, bypassed := 0, 0
|
||||
|
||||
for trial := range 30000 {
|
||||
var sb strings.Builder
|
||||
for range 1 + rng.Intn(8) {
|
||||
sb.WriteString(parts[rng.Intn(len(parts))])
|
||||
}
|
||||
chars := util.ToChars([]byte(sb.String()))
|
||||
if chars.IsBytes() {
|
||||
continue
|
||||
}
|
||||
if chars.MayFoldToAscii() {
|
||||
bypassed++
|
||||
} else {
|
||||
engaged++
|
||||
}
|
||||
pat := patterns[rng.Intn(len(patterns))]
|
||||
cs := rng.Intn(2) == 0
|
||||
if !cs {
|
||||
pat = strings.ToLower(pat)
|
||||
}
|
||||
pattern := []rune(pat)
|
||||
norm := rng.Intn(2) == 0
|
||||
fwd := rng.Intn(2) == 0
|
||||
wp := rng.Intn(2) == 0
|
||||
|
||||
disableRunePrefilter = true
|
||||
expR, expP := FuzzyMatchV2(cs, norm, fwd, &chars, pattern, wp, slab)
|
||||
disableRunePrefilter = false
|
||||
gotR, gotP := FuzzyMatchV2(cs, norm, fwd, &chars, pattern, wp, slab)
|
||||
|
||||
if gotR != expR || !samePos(gotP, expP) {
|
||||
t.Fatalf("trial %d: %q pattern=%q cs=%v norm=%v fwd=%v wp=%v\n prefilter on: %v %v\n prefilter off: %v %v",
|
||||
trial, sb.String(), pat, cs, norm, fwd, wp, gotR, gotP, expR, expP)
|
||||
}
|
||||
}
|
||||
disableRunePrefilter = false
|
||||
t.Logf("prefilter engaged on %d items, bypassed on %d", engaged, bypassed)
|
||||
if engaged == 0 || bypassed == 0 {
|
||||
t.Fatalf("corpus did not exercise both paths (engaged=%d bypassed=%d)", engaged, bypassed)
|
||||
}
|
||||
|
||||
// Equivalence alone would still hold if the prefilter never filtered
|
||||
// anything, so confirm it both rejects and narrows.
|
||||
rejected, narrowed := 0, 0
|
||||
for range 5000 {
|
||||
var sb strings.Builder
|
||||
for range 1 + rng.Intn(8) {
|
||||
sb.WriteString(parts[rng.Intn(len(parts))])
|
||||
}
|
||||
chars := util.ToChars([]byte(sb.String()))
|
||||
if chars.IsBytes() || chars.MayFoldToAscii() {
|
||||
continue
|
||||
}
|
||||
pattern := []rune(patterns[rng.Intn(len(patterns))])
|
||||
lo, hi := asciiFuzzyIndex(&chars, pattern, false)
|
||||
switch {
|
||||
case lo < 0:
|
||||
rejected++
|
||||
case hi-lo < chars.Length():
|
||||
narrowed++
|
||||
}
|
||||
}
|
||||
t.Logf("prefilter rejected %d items, narrowed scope on %d", rejected, narrowed)
|
||||
if rejected == 0 {
|
||||
t.Fatal("prefilter never rejected an item, so equivalence proves nothing")
|
||||
}
|
||||
if narrowed == 0 {
|
||||
t.Fatal("prefilter never narrowed the scope")
|
||||
}
|
||||
}
|
||||
|
||||
// FuzzyMatchV1 shares asciiFuzzyIndex, so it needs the same guarantee.
|
||||
func TestRunePrefilterEquivalenceV1(t *testing.T) {
|
||||
t.Cleanup(func() { disableRunePrefilter = false })
|
||||
rng := rand.New(rand.NewSource(5))
|
||||
parts := []string{"src", "conf", "a", "e", "/", "漢字", "한글", "мир", "café", "Å", "🎉"}
|
||||
slab := util.MakeSlab(100*1024, 2048)
|
||||
for trial := range 20000 {
|
||||
var sb strings.Builder
|
||||
for range 1 + rng.Intn(6) {
|
||||
sb.WriteString(parts[rng.Intn(len(parts))])
|
||||
}
|
||||
chars := util.ToChars([]byte(sb.String()))
|
||||
if chars.IsBytes() {
|
||||
continue
|
||||
}
|
||||
pattern := []rune([]string{"a", "e", "conf", "src", "ae", "zz"}[rng.Intn(6)])
|
||||
cs, norm, fwd, wp := rng.Intn(2) == 0, rng.Intn(2) == 0, rng.Intn(2) == 0, rng.Intn(2) == 0
|
||||
|
||||
disableRunePrefilter = true
|
||||
expR, expP := FuzzyMatchV1(cs, norm, fwd, &chars, pattern, wp, slab)
|
||||
disableRunePrefilter = false
|
||||
gotR, gotP := FuzzyMatchV1(cs, norm, fwd, &chars, pattern, wp, slab)
|
||||
|
||||
if gotR != expR || !samePos(gotP, expP) {
|
||||
t.Fatalf("trial %d: %q pattern=%q\n prefilter on: %v %v\n prefilter off: %v %v",
|
||||
trial, sb.String(), string(pattern), gotR, gotP, expR, expP)
|
||||
}
|
||||
}
|
||||
disableRunePrefilter = false
|
||||
}
|
||||
|
||||
// preparePattern mirrors what pattern.go guarantees the algo functions:
|
||||
// lowercased when case-insensitive, normalized when normalize is on.
|
||||
func preparePattern(pat string, caseSensitive, normalize bool) []rune {
|
||||
if !caseSensitive {
|
||||
pat = strings.ToLower(pat)
|
||||
}
|
||||
r := []rune(pat)
|
||||
if normalize {
|
||||
r = NormalizeRunes(r)
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
// FuzzRunePrefilter drives arbitrary rune-mode input and arbitrary patterns
|
||||
// through the prefilter and through the same code with it disabled, and
|
||||
// requires identical Results and positions. The existing fast-path fuzzers
|
||||
// only generate byte-mode input, so they never reach this path.
|
||||
func FuzzRunePrefilter(f *testing.F) {
|
||||
for _, in := range []string{
|
||||
"한글/src/util.go", "漢字/conf", "café/binutils", "мир/test", "🎉/a",
|
||||
"ABC.txt", "Ångström", "ǰ/ß/İ", "a漢b한c", "Āā",
|
||||
} {
|
||||
for _, p := range []string{"a", "conf", "漢", "한글", "мир", "ß", "É", "a漢"} {
|
||||
f.Add(in, p)
|
||||
}
|
||||
}
|
||||
slab := util.MakeSlab(200*1024, 4096)
|
||||
f.Fuzz(func(t *testing.T, input, pat string) {
|
||||
if len(input) > 512 || len(pat) == 0 || len(pat) > 32 {
|
||||
return
|
||||
}
|
||||
chars := util.ToChars([]byte(input))
|
||||
if chars.IsBytes() {
|
||||
return // byte mode is the existing fuzzers' territory
|
||||
}
|
||||
for _, cs := range []bool{false, true} {
|
||||
for _, norm := range []bool{false, true} {
|
||||
p := preparePattern(pat, cs, norm)
|
||||
if len(p) == 0 {
|
||||
continue
|
||||
}
|
||||
for _, fwd := range []bool{true, false} {
|
||||
for _, wp := range []bool{false, true} {
|
||||
for _, fn := range []Algo{FuzzyMatchV2, FuzzyMatchV1, ExactMatchNaive} {
|
||||
disableRunePrefilter = true
|
||||
expR, expP := fn(cs, norm, fwd, &chars, p, wp, slab)
|
||||
disableRunePrefilter = false
|
||||
gotR, gotP := fn(cs, norm, fwd, &chars, p, wp, slab)
|
||||
if gotR != expR || !samePos(gotP, expP) {
|
||||
t.Fatalf("input=%q pattern=%q cs=%v norm=%v fwd=%v wp=%v\n prefilter on: %v %v\n prefilter off: %v %v",
|
||||
input, pat, cs, norm, fwd, wp, gotR, gotP, expR, expP)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// RunesToChars can produce rune-mode Chars holding zero runes, which sends a
|
||||
// nil pointer through unsafe.SliceData in runeBytes. ToChars cannot produce
|
||||
// this (an empty input is byte mode), so it needs its own test.
|
||||
func TestEmptyRuneModeChars(t *testing.T) {
|
||||
t.Cleanup(func() { disableRunePrefilter = false })
|
||||
slab := util.MakeSlab(100*1024, 2048)
|
||||
for _, runes := range [][]rune{{}, nil, {'a'}, {0x4E00}} {
|
||||
chars := util.RunesToChars(runes)
|
||||
if chars.IsBytes() {
|
||||
continue
|
||||
}
|
||||
for _, pat := range []string{"a", "漢", "ab"} {
|
||||
p := []rune(pat)
|
||||
for _, fn := range []Algo{FuzzyMatchV2, FuzzyMatchV1, ExactMatchNaive,
|
||||
PrefixMatch, SuffixMatch, EqualMatch} {
|
||||
disableRunePrefilter = true
|
||||
expR, expP := fn(false, true, true, &chars, p, true, slab)
|
||||
disableRunePrefilter = false
|
||||
gotR, gotP := fn(false, true, true, &chars, p, true, slab)
|
||||
if gotR != expR || !samePos(gotP, expP) {
|
||||
t.Errorf("runes=%U pat=%q: prefilter on %v %v, off %v %v", runes, pat, gotR, gotP, expR, expP)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// MayFoldToAscii subtracts before bounds-checking, so a rune below the range
|
||||
// (including a negative one, which utf8 decoding never produces but callers
|
||||
// could construct) must not wrap into a false positive.
|
||||
func TestMayFoldToAsciiOutOfRange(t *testing.T) {
|
||||
for _, r := range []rune{-1, -0x10000, 0, 'a', 0x7F, 0xBF, 0xFF62, unicode.MaxRune, unicode.MaxRune + 1} {
|
||||
if util.MayFoldToAscii(r) {
|
||||
t.Errorf("MayFoldToAscii(%d) = true, expected false", r)
|
||||
}
|
||||
}
|
||||
}
|
||||
+108
-8
@@ -14,9 +14,19 @@ const (
|
||||
overflow32 uint32 = 0x80808080
|
||||
)
|
||||
|
||||
const (
|
||||
flagInBytes uint8 = 1 << iota
|
||||
flagMayFold
|
||||
)
|
||||
|
||||
type Chars struct {
|
||||
slice []byte // or []rune
|
||||
inBytes bool
|
||||
slice []byte // or []rune
|
||||
// Written while the item is built and never after it reaches a matcher,
|
||||
// so nothing reads these bits concurrently with a write. Prepend touches
|
||||
// them, but only on the transient tokens inside transformItem, before
|
||||
// item.text exists. trimLength* is kept out because TrimLength writes it
|
||||
// lazily, long after that point.
|
||||
flags uint8
|
||||
trimLengthKnown bool
|
||||
trimLength uint16
|
||||
|
||||
@@ -25,6 +35,59 @@ type Chars struct {
|
||||
Index int32
|
||||
}
|
||||
|
||||
// Rune ranges that case folding or normalization can turn into ASCII, derived
|
||||
// from algo's normalization table and unicode.ToLower, then merged. They are a
|
||||
// superset of the exact set, which TestMayFoldToAsciiIsSuperset in the algo
|
||||
// package pins. Grouped tightly on purpose: a wider merge would swallow Greek
|
||||
// Extended, General Punctuation and the currency and letterlike blocks, and
|
||||
// every line holding a curly quote or an em dash would then lose the
|
||||
// prefilter. Cyrillic, Greek, Hebrew, Arabic, Thai, Devanagari, CJK, Hangul,
|
||||
// kana, emoji, punctuation and box drawing are all outside.
|
||||
const (
|
||||
foldLo = 0x00C0
|
||||
foldHi = 0xFF61
|
||||
)
|
||||
|
||||
var foldableRanges = [...][2]rune{
|
||||
{0x00C0, 0x01B6}, // Latin-1 Supplement, Latin Extended-A and -B
|
||||
{0x01CD, 0x02AE}, // rest of Latin Extended-B and IPA Extensions
|
||||
{0x0363, 0x036F}, // combining Latin small letters
|
||||
{0x1D00, 0x1D22}, // Phonetic Extensions, small capitals
|
||||
{0x1D62, 0x1D65}, // subscript letters
|
||||
{0x1E00, 0x1EF9}, // Latin Extended Additional
|
||||
{0x2071, 0x2071}, // superscript i
|
||||
{0x2095, 0x209C}, // subscript letters
|
||||
{0x212A, 0x212B}, // KELVIN SIGN and ANGSTROM SIGN, which fold by case
|
||||
{0x2183, 0x2184}, // reversed roman numeral one hundred
|
||||
{0x2C62, 0x2C7F}, // Latin Extended-C
|
||||
{0xA78D, 0xA78D}, // Latin Extended-D
|
||||
{0xA7AA, 0xA7B2}, // more Latin Extended-D
|
||||
{0xA7C5, 0xA7C5},
|
||||
{0xFF01, 0xFF61}, // fullwidth ASCII forms, and halfwidth ideographic full stop
|
||||
}
|
||||
|
||||
// Walking the ranges costs a serial chain of comparisons per rune, which is
|
||||
// measurable at ingestion, so precompute a bitmap instead.
|
||||
var foldableBits = func() (bits [(foldHi-foldLo)/8 + 1]byte) {
|
||||
for _, r := range foldableRanges {
|
||||
for c := r[0]; c <= r[1]; c++ {
|
||||
i := c - foldLo
|
||||
bits[i>>3] |= 1 << (i & 7)
|
||||
}
|
||||
}
|
||||
return
|
||||
}()
|
||||
|
||||
// MayFoldToAscii reports whether case folding or normalization could turn r
|
||||
// into an ASCII character.
|
||||
func MayFoldToAscii(r rune) bool {
|
||||
i := uint32(r - foldLo)
|
||||
if i > foldHi-foldLo {
|
||||
return false
|
||||
}
|
||||
return foldableBits[i>>3]&(1<<(i&7)) != 0
|
||||
}
|
||||
|
||||
func checkAscii(bytes []byte) (bool, int) {
|
||||
i := 0
|
||||
for ; i <= len(bytes)-8; i += 8 {
|
||||
@@ -69,16 +132,18 @@ func countRunes(bytes []byte) int {
|
||||
func ToChars(bytes []byte) Chars {
|
||||
inBytes, bytesUntil := checkAscii(bytes)
|
||||
if inBytes {
|
||||
return Chars{slice: bytes, inBytes: inBytes}
|
||||
return Chars{slice: bytes, flags: flagInBytes}
|
||||
}
|
||||
|
||||
runes := make([]rune, bytesUntil, bytesUntil+countRunes(bytes[bytesUntil:]))
|
||||
for i := range bytesUntil {
|
||||
runes[i] = rune(bytes[i])
|
||||
}
|
||||
mayFold := false
|
||||
for i := bytesUntil; i < len(bytes); {
|
||||
// utf8.DecodeRune has an ASCII path of its own, but it is too complex
|
||||
// to inline, so a mostly-ASCII line pays one call per byte for it.
|
||||
// An ASCII rune never sets the fold bit either, so skip both calls.
|
||||
if b := bytes[i]; b < utf8.RuneSelf {
|
||||
runes = append(runes, rune(b))
|
||||
i++
|
||||
@@ -86,17 +151,46 @@ func ToChars(bytes []byte) Chars {
|
||||
}
|
||||
r, sz := utf8.DecodeRune(bytes[i:])
|
||||
i += sz
|
||||
mayFold = mayFold || MayFoldToAscii(r)
|
||||
runes = append(runes, r)
|
||||
}
|
||||
return RunesToChars(runes)
|
||||
return runesToChars(runes, mayFold)
|
||||
}
|
||||
|
||||
func RunesToChars(runes []rune) Chars {
|
||||
return Chars{slice: *(*[]byte)(unsafe.Pointer(&runes)), inBytes: false}
|
||||
mayFold := false
|
||||
for _, r := range runes {
|
||||
if MayFoldToAscii(r) {
|
||||
mayFold = true
|
||||
break
|
||||
}
|
||||
}
|
||||
return runesToChars(runes, mayFold)
|
||||
}
|
||||
|
||||
func runesToChars(runes []rune, mayFold bool) Chars {
|
||||
var flags uint8
|
||||
if mayFold {
|
||||
flags = flagMayFold
|
||||
}
|
||||
return Chars{slice: *(*[]byte)(unsafe.Pointer(&runes)), flags: flags}
|
||||
}
|
||||
|
||||
func (chars *Chars) IsBytes() bool {
|
||||
return chars.inBytes
|
||||
return chars.flags&flagInBytes != 0
|
||||
}
|
||||
|
||||
// MayFoldToAscii reports whether the text holds a rune that case folding or
|
||||
// normalization could turn into an ASCII character. When false, an ASCII
|
||||
// pattern character can only match the identical ASCII rune, which is what
|
||||
// lets the prefilter scan the rune array directly.
|
||||
func (chars *Chars) MayFoldToAscii() bool {
|
||||
return chars.flags&flagMayFold != 0
|
||||
}
|
||||
|
||||
// Runes returns the underlying rune slice, or nil if the text is kept as bytes.
|
||||
func (chars *Chars) Runes() []rune {
|
||||
return chars.optionalRunes()
|
||||
}
|
||||
|
||||
func (chars *Chars) Bytes() []byte {
|
||||
@@ -133,7 +227,7 @@ func (chars *Chars) NumLines(atMost int) (int, bool) {
|
||||
}
|
||||
|
||||
func (chars *Chars) optionalRunes() []rune {
|
||||
if chars.inBytes {
|
||||
if chars.IsBytes() {
|
||||
return nil
|
||||
}
|
||||
return *(*[]rune)(unsafe.Pointer(&chars.slice))
|
||||
@@ -155,7 +249,7 @@ func (chars *Chars) Length() int {
|
||||
|
||||
// String returns the string representation of a Chars object.
|
||||
func (chars *Chars) String() string {
|
||||
return fmt.Sprintf("Chars{slice: []byte(%q), inBytes: %v, trimLengthKnown: %v, trimLength: %d, Index: %d}", chars.slice, chars.inBytes, chars.trimLengthKnown, chars.trimLength, chars.Index)
|
||||
return fmt.Sprintf("Chars{slice: []byte(%q), inBytes: %v, mayFold: %v, trimLengthKnown: %v, trimLength: %d, Index: %d}", chars.slice, chars.IsBytes(), chars.MayFoldToAscii(), chars.trimLengthKnown, chars.trimLength, chars.Index)
|
||||
}
|
||||
|
||||
// TrimLength returns the length after trimming leading and trailing whitespaces
|
||||
@@ -275,6 +369,12 @@ func (chars *Chars) Prepend(prefix string) {
|
||||
} else {
|
||||
chars.slice = append([]byte(prefix), chars.slice...)
|
||||
}
|
||||
for _, r := range prefix {
|
||||
if MayFoldToAscii(r) {
|
||||
chars.flags |= flagMayFold
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (chars *Chars) Lines(multiLine bool, maxLines int, wrapCols int, wrapSignWidth int, tabstop int, wrapWord bool) ([][]rune, bool) {
|
||||
|
||||
+44
-2
@@ -6,6 +6,7 @@ import (
|
||||
"strings"
|
||||
"testing"
|
||||
"unicode/utf8"
|
||||
"unsafe"
|
||||
)
|
||||
|
||||
func TestCountRunes(t *testing.T) {
|
||||
@@ -93,14 +94,14 @@ func TestToCharsIntegrity(t *testing.T) {
|
||||
|
||||
func TestToCharsAscii(t *testing.T) {
|
||||
chars := ToChars([]byte("foobar"))
|
||||
if !chars.inBytes || chars.ToString() != "foobar" || !chars.inBytes {
|
||||
if !chars.IsBytes() || chars.ToString() != "foobar" {
|
||||
t.Error()
|
||||
}
|
||||
}
|
||||
|
||||
func TestCharsLength(t *testing.T) {
|
||||
chars := ToChars([]byte("\tabc한글 "))
|
||||
if chars.inBytes || chars.Length() != 8 || chars.TrimLength() != 5 {
|
||||
if chars.IsBytes() || chars.Length() != 8 || chars.TrimLength() != 5 {
|
||||
t.Error()
|
||||
}
|
||||
}
|
||||
@@ -215,3 +216,44 @@ func TestCharsLinesWrapWord(t *testing.T) {
|
||||
t.Errorf("Expected first line 'hello wo', got %q", string(lines4[0]))
|
||||
}
|
||||
}
|
||||
|
||||
// Chars is one per input line, so its size is load-bearing. It has no spare
|
||||
// padding, which is why new state goes in the flags byte rather than a field.
|
||||
// Derive the expectation from the slice header so the invariant holds on
|
||||
// 32-bit builds too, where the header is 12 bytes and Chars is 20.
|
||||
func TestCharsSize(t *testing.T) {
|
||||
var slice []byte
|
||||
// flags 1 + trimLengthKnown 1 + trimLength 2 + Index 4, no padding
|
||||
want := unsafe.Sizeof(slice) + 8
|
||||
if size := unsafe.Sizeof(Chars{}); size != want {
|
||||
t.Errorf("unsafe.Sizeof(Chars{}) = %d, expected %d", size, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMayFoldFlag(t *testing.T) {
|
||||
for _, c := range []struct {
|
||||
text string
|
||||
fold bool
|
||||
}{
|
||||
{"한글/src", false}, {"漢字", false}, {"мир", false}, {"🎉", false},
|
||||
{"café", true}, {"Müller", true}, {"Å", true}, {"full", true},
|
||||
} {
|
||||
chars := ToChars([]byte(c.text))
|
||||
if chars.MayFoldToAscii() != c.fold {
|
||||
t.Errorf("ToChars(%q).MayFoldToAscii() = %v, expected %v", c.text, chars.MayFoldToAscii(), c.fold)
|
||||
}
|
||||
if runes := RunesToChars([]rune(c.text)); runes.MayFoldToAscii() != c.fold {
|
||||
t.Errorf("RunesToChars(%q).MayFoldToAscii() = %v, expected %v", c.text, runes.MayFoldToAscii(), c.fold)
|
||||
}
|
||||
}
|
||||
|
||||
// Prepend can introduce foldable runes
|
||||
chars := ToChars([]byte("한글"))
|
||||
if chars.MayFoldToAscii() {
|
||||
t.Fatal("baseline should not be foldable")
|
||||
}
|
||||
chars.Prepend("é")
|
||||
if !chars.MayFoldToAscii() {
|
||||
t.Error("Prepend of a foldable prefix must set the flag")
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user