Merge branch 'main' into v2
* main: chore(fs): speed up case normalization (#10013) chore(config): remove discontinued secondary STUN servers (fixes #10011) (#10012) chore(gui, man, authors): update docs, translations, and contributors fix(stun): better error handling (ref #10008) (#10010) fix(config): remove discontinued primary STUN server (fixes #10008) (#10009) fix(gui): validate device ID in canonical form (fixes #7291) (#10006)
This commit is contained in:
+57
-4
@@ -17,6 +17,52 @@ import (
|
||||
// UnicodeLowercaseNormalized returns the Unicode lower case variant of s,
|
||||
// having also normalized it to normalization form C.
|
||||
func UnicodeLowercaseNormalized(s string) string {
|
||||
if isASCII, isLower := isASCII(s); isASCII {
|
||||
if isLower {
|
||||
return s
|
||||
}
|
||||
return toLowerASCII(s)
|
||||
}
|
||||
|
||||
return toLowerUnicode(s)
|
||||
}
|
||||
|
||||
func isASCII(s string) (bool, bool) {
|
||||
isLower := true
|
||||
for _, b := range []byte(s) {
|
||||
if b > unicode.MaxASCII {
|
||||
return false, isLower
|
||||
}
|
||||
if 'A' <= b && b <= 'Z' {
|
||||
isLower = false
|
||||
}
|
||||
}
|
||||
return true, isLower
|
||||
}
|
||||
|
||||
func toLowerASCII(s string) string {
|
||||
var (
|
||||
b strings.Builder
|
||||
pos int
|
||||
)
|
||||
b.Grow(len(s))
|
||||
for i, c := range []byte(s) {
|
||||
if c < 'A' || 'Z' < c {
|
||||
continue
|
||||
}
|
||||
if pos < i {
|
||||
b.WriteString(s[pos:i])
|
||||
}
|
||||
pos = i + 1
|
||||
b.WriteByte(c + 'a' - 'A')
|
||||
}
|
||||
if pos != len(s) {
|
||||
b.WriteString(s[pos:])
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func toLowerUnicode(s string) string {
|
||||
i := firstCaseChange(s)
|
||||
if i == -1 {
|
||||
return norm.NFC.String(s)
|
||||
@@ -30,7 +76,11 @@ func UnicodeLowercaseNormalized(s string) string {
|
||||
rs.WriteString(s[:i])
|
||||
|
||||
for _, r := range s[i:] {
|
||||
rs.WriteRune(unicode.ToLower(unicode.ToUpper(r)))
|
||||
if r <= unicode.MaxLatin1 && r != 'µ' {
|
||||
rs.WriteRune(unicode.ToLower(r))
|
||||
} else {
|
||||
rs.WriteRune(unicode.To(unicode.LowerCase, unicode.To(unicode.UpperCase, r)))
|
||||
}
|
||||
}
|
||||
return norm.NFC.String(rs.String())
|
||||
}
|
||||
@@ -38,10 +88,13 @@ func UnicodeLowercaseNormalized(s string) string {
|
||||
// Byte index of the first rune r s.t. lower(upper(r)) != r.
|
||||
func firstCaseChange(s string) int {
|
||||
for i, r := range s {
|
||||
if r <= unicode.MaxASCII && (r < 'A' || r > 'Z') {
|
||||
continue
|
||||
if r <= unicode.MaxASCII {
|
||||
if r < 'A' || r > 'Z' {
|
||||
continue
|
||||
}
|
||||
return i
|
||||
}
|
||||
if unicode.ToLower(unicode.ToUpper(r)) != r {
|
||||
if unicode.To(unicode.LowerCase, unicode.To(unicode.UpperCase, r)) != r {
|
||||
return i
|
||||
}
|
||||
}
|
||||
|
||||
+20
-17
@@ -49,6 +49,18 @@ var caseCases = [][2]string{
|
||||
{"a\xCC\x88", "\xC3\xA4"}, // ä
|
||||
}
|
||||
|
||||
var benchmarkCases = [][2]string{
|
||||
{"img_202401241010.jpg", "ASCII lowercase"},
|
||||
{"IMG_202401241010.jpg", "ASCII mixedcase start"},
|
||||
{"img_202401241010.JPG", "ASCII mixedcase end"},
|
||||
{"wir_kinder_aus_bullerbü.epub", "Latin1 lowercase"},
|
||||
{"Wir_Kinder_aus_Bullerbü.epub", "Latin1 mixedcase start"},
|
||||
{"wir_kinder_aus_bullerbü.EPUB", "Latin1 mixedcase end"},
|
||||
{"translated_ウェブの国際化.html", "Unicode lowercase"},
|
||||
{"Translated_ウェブの国際化.html", "Unicode mixedcase start"},
|
||||
{"translated_ウェブの国際化.HTML", "Unicode mixedcase end"},
|
||||
}
|
||||
|
||||
func TestUnicodeLowercaseNormalized(t *testing.T) {
|
||||
for _, tc := range caseCases {
|
||||
res := UnicodeLowercaseNormalized(tc[0])
|
||||
@@ -58,22 +70,13 @@ func TestUnicodeLowercaseNormalized(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkUnicodeLowercaseMaybeChange(b *testing.B) {
|
||||
b.ReportAllocs()
|
||||
|
||||
for i := 0; i < b.N; i++ {
|
||||
for _, s := range caseCases {
|
||||
UnicodeLowercaseNormalized(s[0])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkUnicodeLowercaseNoChange(b *testing.B) {
|
||||
b.ReportAllocs()
|
||||
|
||||
for i := 0; i < b.N; i++ {
|
||||
for _, s := range caseCases {
|
||||
UnicodeLowercaseNormalized(s[1])
|
||||
}
|
||||
func BenchmarkUnicodeLowercase(b *testing.B) {
|
||||
for _, c := range benchmarkCases {
|
||||
b.Run(c[1], func(b *testing.B) {
|
||||
b.ReportAllocs()
|
||||
for i := 0; i < b.N; i++ {
|
||||
UnicodeLowercaseNormalized(c[0])
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user