70 lines
2.0 KiB
Go
70 lines
2.0 KiB
Go
// Copyright 2023 Dolthub, Inc.
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
|
|
// Copyright 2017 The Cockroach Authors.
|
|
//
|
|
// Use of this software is governed by the Business Source License
|
|
// included in the file licenses/BSL.txt.
|
|
//
|
|
// As of the Change Date specified in that file, in accordance with
|
|
// the Business Source License, use of this software will be governed
|
|
// by the Apache License, Version 2.0, included in the file
|
|
// licenses/APL.txt.
|
|
|
|
package lex
|
|
|
|
import (
|
|
"strings"
|
|
"unicode"
|
|
|
|
"golang.org/x/text/unicode/norm"
|
|
)
|
|
|
|
// Special case normalization rules for Turkish/Azeri lowercase dotless-i and
|
|
// uppercase dotted-i. Fold both dotted and dotless 'i' into the ascii i/I, so
|
|
// our case-insensitive comparison functions can be locale-invariant. This
|
|
// mapping implements case-insensitivity for Turkish and other latin-derived
|
|
// languages simultaneously, with the additional quirk that it is also
|
|
// insensitive to the dottedness of the i's
|
|
var normalize = unicode.SpecialCase{
|
|
unicode.CaseRange{
|
|
Lo: 0x0130,
|
|
Hi: 0x0130,
|
|
Delta: [unicode.MaxCase]rune{
|
|
0x49 - 0x130, // Upper
|
|
0x69 - 0x130, // Lower
|
|
0x49 - 0x130, // Title
|
|
},
|
|
},
|
|
unicode.CaseRange{
|
|
Lo: 0x0131,
|
|
Hi: 0x0131,
|
|
Delta: [unicode.MaxCase]rune{
|
|
0x49 - 0x131, // Upper
|
|
0x69 - 0x131, // Lower
|
|
0x49 - 0x131, // Title
|
|
},
|
|
},
|
|
}
|
|
|
|
// NormalizeName normalizes to lowercase and Unicode Normalization
|
|
// Form C (NFC).
|
|
func NormalizeName(n string) string {
|
|
lower := strings.Map(normalize.ToLower, n)
|
|
if isASCII(lower) {
|
|
return lower
|
|
}
|
|
return norm.NFC.String(lower)
|
|
}
|