// SPDX-License-Identifier: GPL-3.0-or-later // Package norm puts text into the form krino compares it in: optionally // without diacritics, optionally lower case, with white space collapsed. package norm import ( "fmt" "strings" "unicode" unorm "golang.org/x/text/unicode/norm" ) // Version identifies what Text and Name produce. Bump it whenever a change // could make any string normalise differently: keyword answers cached under // another version are discarded (spec §6.1). const Version = 1 // Fingerprint names everything Text and Name's output depends on: Version, // and the Unicode tables of the standard library and of x/text's // normalisation, which a Go or x/text upgrade can change (review cache F4). func Fingerprint() string { return fmt.Sprintf("norm%d unicode%s nfd%s", Version, unicode.Version, unorm.Version) } // special holds the letters that do not decompose under Unicode NFD, so // Fold maps them explicitly. var special = map[rune]string{ 'ł': "l", 'Ł': "L", 'ø': "o", 'Ø': "O", 'đ': "d", 'Đ': "D", 'ħ': "h", 'Ħ': "H", 'ß': "ss", 'ẞ': "SS", 'æ': "ae", 'Æ': "AE", 'œ': "oe", 'Œ': "OE", 'ı': "i", } // Fold strips diacritics: Unicode NFD, drop combining marks (category Mn), // then map the letters that do not decompose. Invalid UTF-8 becomes U+FFFD // first: left in, an incomplete sequence can keep NFD from decomposing the // letter after it, and folding the result again would change it. ASCII // input is returned unchanged without allocating. func Fold(s string) string { ascii := true for i := 0; i < len(s); i++ { if s[i] >= 0x80 { ascii = false break } } if ascii { return s } s = strings.ToValidUTF8(s, "\ufffd") var b strings.Builder b.Grow(len(s)) for _, r := range unorm.NFD.String(s) { if unicode.Is(unicode.Mn, r) { continue } if rep, ok := special[r]; ok { b.WriteString(rep) continue } b.WriteRune(r) } return b.String() } // Text puts s into the form content and keywords are compared in: Fold if // fold, strings.ToLower if ignoreCase, then every run of Unicode white // space becomes one ASCII space and both ends are trimmed. func Text(s string, ignoreCase, fold bool) string { if fold { s = Fold(s) } if ignoreCase { s = strings.ToLower(s) } var b strings.Builder b.Grow(len(s)) inSpace := false started := false for _, r := range s { if unicode.IsSpace(r) { if started { inSpace = true } continue } if inSpace { b.WriteByte(' ') inSpace = false } b.WriteRune(r) started = true } return b.String() } // Name puts a file name into the form it is matched against: Fold(s) if // fold, else s. Case is handled by the regex flag, not here. func Name(s string, fold bool) string { if fold { return Fold(s) } return s }