From 05cc6b0f432f0bfc707ce4dd4c97154593d92d51 Mon Sep 17 00:00:00 2001 From: bt90 Date: Tue, 1 Apr 2025 13:41:57 +0200 Subject: [PATCH] chore(fs): speed up case normalization (#10013) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ### Purpose Resurrecting https://github.com/syncthing/syncthing/pull/9365 ### Testing Current benchmark results: ``` goos: linux goarch: amd64 pkg: github.com/syncthing/syncthing/lib/fs cpu: AMD EPYC 7763 64-Core Processor │ ../old.txt │ ../new.txt │ │ sec/op │ sec/op vs base │ UnicodeLowercase/ASCII_lowercase-4 43.68n ± 4% 12.19n ± 1% -72.09% (p=0.000 n=10) UnicodeLowercase/ASCII_mixedcase_start-4 200.65n ± 2% 59.49n ± 3% -70.35% (p=0.000 n=10) UnicodeLowercase/ASCII_mixedcase_end-4 95.50n ± 2% 59.10n ± 2% -38.12% (p=0.000 n=10) UnicodeLowercase/Latin1_lowercase-4 122.5n ± 1% 131.4n ± 1% +7.27% (p=0.000 n=10) UnicodeLowercase/Latin1_mixedcase_start-4 339.9n ± 2% 309.2n ± 1% -9.05% (p=0.000 n=10) UnicodeLowercase/Latin1_mixedcase_end-4 183.6n ± 2% 174.3n ± 1% -5.04% (p=0.000 n=10) UnicodeLowercase/Unicode_lowercase-4 456.6n ± 1% 440.5n ± 1% -3.53% (p=0.000 n=10) UnicodeLowercase/Unicode_mixedcase_start-4 625.9n ± 1% 595.6n ± 1% -4.83% (p=0.000 n=10) UnicodeLowercase/Unicode_mixedcase_end-4 516.2n ± 1% 495.5n ± 1% -4.02% (p=0.000 n=10) geomean 214.1n 150.4n -29.72% │ ../old.txt │ ../new.txt │ │ B/op │ B/op vs base │ UnicodeLowercase/ASCII_lowercase-4 0.000 ± 0% 0.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/ASCII_mixedcase_start-4 24.00 ± 0% 24.00 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/ASCII_mixedcase_end-4 24.00 ± 0% 24.00 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Latin1_lowercase-4 0.000 ± 0% 0.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Latin1_mixedcase_start-4 32.00 ± 0% 32.00 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Latin1_mixedcase_end-4 32.00 ± 0% 32.00 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Unicode_lowercase-4 0.000 ± 0% 0.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Unicode_mixedcase_start-4 48.00 ± 0% 48.00 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Unicode_mixedcase_end-4 48.00 ± 0% 48.00 ± 0% ~ (p=1.000 n=10) ¹ geomean ² +0.00% ² ¹ all samples are equal ² summaries must be >0 to compute geomean │ ../old.txt │ ../new.txt │ │ allocs/op │ allocs/op vs base │ UnicodeLowercase/ASCII_lowercase-4 0.000 ± 0% 0.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/ASCII_mixedcase_start-4 1.000 ± 0% 1.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/ASCII_mixedcase_end-4 1.000 ± 0% 1.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Latin1_lowercase-4 0.000 ± 0% 0.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Latin1_mixedcase_start-4 1.000 ± 0% 1.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Latin1_mixedcase_end-4 1.000 ± 0% 1.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Unicode_lowercase-4 0.000 ± 0% 0.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Unicode_mixedcase_start-4 1.000 ± 0% 1.000 ± 0% ~ (p=1.000 n=10) ¹ UnicodeLowercase/Unicode_mixedcase_end-4 1.000 ± 0% 1.000 ± 0% ~ (p=1.000 n=10) ¹ geomean ² +0.00% ² ¹ all samples are equal ² summaries must be >0 to compute geomean ``` I think the `+7%` for the lowercase Latin1 testcase is easily outweighed by the ASCII and unicode improvements :slightly_smiling_face: --- lib/fs/folding.go | 61 +++++++++++++++++++++++++++++++++++++++--- lib/fs/folding_test.go | 37 +++++++++++++------------ 2 files changed, 77 insertions(+), 21 deletions(-) diff --git a/lib/fs/folding.go b/lib/fs/folding.go index 20887b95e..1a8791d43 100644 --- a/lib/fs/folding.go +++ b/lib/fs/folding.go @@ -17,6 +17,52 @@ import ( // UnicodeLowercaseNormalized returns the Unicode lower case variant of s, // having also normalized it to normalization form C. func UnicodeLowercaseNormalized(s string) string { + if isASCII, isLower := isASCII(s); isASCII { + if isLower { + return s + } + return toLowerASCII(s) + } + + return toLowerUnicode(s) +} + +func isASCII(s string) (bool, bool) { + isLower := true + for _, b := range []byte(s) { + if b > unicode.MaxASCII { + return false, isLower + } + if 'A' <= b && b <= 'Z' { + isLower = false + } + } + return true, isLower +} + +func toLowerASCII(s string) string { + var ( + b strings.Builder + pos int + ) + b.Grow(len(s)) + for i, c := range []byte(s) { + if c < 'A' || 'Z' < c { + continue + } + if pos < i { + b.WriteString(s[pos:i]) + } + pos = i + 1 + b.WriteByte(c + 'a' - 'A') + } + if pos != len(s) { + b.WriteString(s[pos:]) + } + return b.String() +} + +func toLowerUnicode(s string) string { i := firstCaseChange(s) if i == -1 { return norm.NFC.String(s) @@ -30,7 +76,11 @@ func UnicodeLowercaseNormalized(s string) string { rs.WriteString(s[:i]) for _, r := range s[i:] { - rs.WriteRune(unicode.ToLower(unicode.ToUpper(r))) + if r <= unicode.MaxLatin1 && r != 'µ' { + rs.WriteRune(unicode.ToLower(r)) + } else { + rs.WriteRune(unicode.To(unicode.LowerCase, unicode.To(unicode.UpperCase, r))) + } } return norm.NFC.String(rs.String()) } @@ -38,10 +88,13 @@ func UnicodeLowercaseNormalized(s string) string { // Byte index of the first rune r s.t. lower(upper(r)) != r. func firstCaseChange(s string) int { for i, r := range s { - if r <= unicode.MaxASCII && (r < 'A' || r > 'Z') { - continue + if r <= unicode.MaxASCII { + if r < 'A' || r > 'Z' { + continue + } + return i } - if unicode.ToLower(unicode.ToUpper(r)) != r { + if unicode.To(unicode.LowerCase, unicode.To(unicode.UpperCase, r)) != r { return i } } diff --git a/lib/fs/folding_test.go b/lib/fs/folding_test.go index c3e05fa8b..fd944047b 100644 --- a/lib/fs/folding_test.go +++ b/lib/fs/folding_test.go @@ -49,6 +49,18 @@ var caseCases = [][2]string{ {"a\xCC\x88", "\xC3\xA4"}, // ä } +var benchmarkCases = [][2]string{ + {"img_202401241010.jpg", "ASCII lowercase"}, + {"IMG_202401241010.jpg", "ASCII mixedcase start"}, + {"img_202401241010.JPG", "ASCII mixedcase end"}, + {"wir_kinder_aus_bullerbü.epub", "Latin1 lowercase"}, + {"Wir_Kinder_aus_Bullerbü.epub", "Latin1 mixedcase start"}, + {"wir_kinder_aus_bullerbü.EPUB", "Latin1 mixedcase end"}, + {"translated_ウェブの国際化.html", "Unicode lowercase"}, + {"Translated_ウェブの国際化.html", "Unicode mixedcase start"}, + {"translated_ウェブの国際化.HTML", "Unicode mixedcase end"}, +} + func TestUnicodeLowercaseNormalized(t *testing.T) { for _, tc := range caseCases { res := UnicodeLowercaseNormalized(tc[0]) @@ -58,22 +70,13 @@ func TestUnicodeLowercaseNormalized(t *testing.T) { } } -func BenchmarkUnicodeLowercaseMaybeChange(b *testing.B) { - b.ReportAllocs() - - for i := 0; i < b.N; i++ { - for _, s := range caseCases { - UnicodeLowercaseNormalized(s[0]) - } - } -} - -func BenchmarkUnicodeLowercaseNoChange(b *testing.B) { - b.ReportAllocs() - - for i := 0; i < b.N; i++ { - for _, s := range caseCases { - UnicodeLowercaseNormalized(s[1]) - } +func BenchmarkUnicodeLowercase(b *testing.B) { + for _, c := range benchmarkCases { + b.Run(c[1], func(b *testing.B) { + b.ReportAllocs() + for i := 0; i < b.N; i++ { + UnicodeLowercaseNormalized(c[0]) + } + }) } }