feat: extract normalization into internal/normalize package

This commit is contained in:
2026-07-19 18:11:22 +03:00
parent c95c740cd5
commit edf9c1d1f8
4 changed files with 169 additions and 68 deletions

View File

@@ -0,0 +1,80 @@
// Package normalize provides string normalization helpers used for
// fuzzy matching across NaviWatcher (artist names, album titles, etc.).
//
// It is the single shared home for normalization logic; previously this
// lived inside the musicbrainz package but is needed by the scanner engine
// and any other consumer that compares strings.
package normalize
import (
"regexp"
"strings"
"unicode"
)
// Precompiled regexes — compiled once at package init.
var (
bracketRe = regexp.MustCompile(`\[[^\]]*\]`)
parenRe = regexp.MustCompile(`\([^)]*\)`)
yearRe = regexp.MustCompile(`\b(1[0-9]{3}|2[0-9]{3})\b`)
spaceRe = regexp.MustCompile(`\s+`)
)
// NormalizeString normalizes a string for fuzzy matching by:
// - Converting to lowercase
// - Removing special characters (keeping only letters, digits, and spaces)
// - Removing years (4-digit numbers that look like years)
// - Removing bracketed keywords (e.g., [Deluxe], [Remastered])
// - Collapsing multiple spaces into one
// - Trimming leading/trailing whitespace
func NormalizeString(s string) string {
// Convert to lowercase
s = strings.ToLower(s)
// Remove bracketed content first (e.g., [Deluxe Edition], [Remastered 2020])
s = bracketRe.ReplaceAllString(s, "")
// Remove parenthesized content (e.g., (Deluxe), (Remastered))
s = parenRe.ReplaceAllString(s, "")
// Remove years (4-digit numbers between 1000-2999)
s = yearRe.ReplaceAllString(s, "")
// Replace common separators with spaces before stripping other special chars
s = strings.ReplaceAll(s, "-", " ")
s = strings.ReplaceAll(s, "_", " ")
// Keep only letters, digits, and spaces
var b strings.Builder
for _, r := range s {
if unicode.IsLetter(r) || unicode.IsDigit(r) || unicode.IsSpace(r) {
b.WriteRune(r)
}
}
s = b.String()
// Collapse multiple spaces
s = spaceRe.ReplaceAllString(s, " ")
// Trim
s = strings.TrimSpace(s)
return s
}
// NormalizeArtistName normalizes an artist name for comparison.
// It applies NormalizeString and additionally handles common prefixes.
func NormalizeArtistName(name string) string {
name = NormalizeString(name)
// Remove common leading articles for better matching
prefixes := []string{"the ", "a ", "an "}
for _, prefix := range prefixes {
if strings.HasPrefix(name, prefix) {
name = strings.TrimPrefix(name, prefix)
break
}
}
return strings.TrimSpace(name)
}

View File

@@ -0,0 +1,76 @@
package normalize
import "testing"
func TestNormalizeString_Basic(t *testing.T) {
tests := []struct {
input string
expected string
}{
// Lowercase conversion
{"DARK SIDE OF THE MOON", "dark side of the moon"},
// Special character removal
{"Dark Side of the Moon!", "dark side of the moon"},
{"Dark-Side-of-the-Moon", "dark side of the moon"},
{"Dark_Side_of_the_Moon", "dark side of the moon"},
// Bracket removal
{"Dark Side of the Moon [Deluxe Edition]", "dark side of the moon"},
{"Dark Side of the Moon [Remastered 2020]", "dark side of the moon"},
{"Album [2023 Remix]", "album"},
// Parenthesis removal
{"Dark Side of the Moon (Deluxe)", "dark side of the moon"},
{"Album (Remastered)", "album"},
// Year removal
{"Dark Side of the Moon 1973", "dark side of the moon"},
{"Album 2020 Remastered", "album remastered"},
// Space collapsing
{"Dark Side of the Moon", "dark side of the moon"},
// Trim
{" Dark Side of the Moon ", "dark side of the moon"},
// Combined
{"The Dark Side of the Moon [2011 Remaster] (Deluxe Edition)", "the dark side of the moon"},
// Empty
{"", ""},
// Only special chars
{"!@#$%^&*()", ""},
// Digits that are not years should stay
{"30 Seconds to Mars", "30 seconds to mars"},
{"1941 - The Greatest Hits", "the greatest hits"},
}
for _, tt := range tests {
t.Run(tt.input, func(t *testing.T) {
got := NormalizeString(tt.input)
if got != tt.expected {
t.Errorf("NormalizeString(%q) = %q, want %q", tt.input, got, tt.expected)
}
})
}
}
func TestNormalizeArtistName(t *testing.T) {
tests := []struct {
input string
expected string
}{
{"Pink Floyd", "pink floyd"},
{"The Beatles", "beatles"},
{"A Perfect Circle", "perfect circle"},
{"An Orchestra", "orchestra"},
{" The Who ", "who"},
{"THE WHO", "who"},
// No stripping needed
{"Radiohead", "radiohead"},
// Already stripped
{"Beatles", "beatles"},
}
for _, tt := range tests {
t.Run(tt.input, func(t *testing.T) {
got := NormalizeArtistName(tt.input)
if got != tt.expected {
t.Errorf("NormalizeArtistName(%q) = %q, want %q", tt.input, got, tt.expected)
}
})
}
}