diff --git a/internal/catalog/item_repo.go b/internal/catalog/item_repo.go index 3e921b1e..f398c006 100644 --- a/internal/catalog/item_repo.go +++ b/internal/catalog/item_repo.go @@ -812,11 +812,10 @@ func (r *ItemRepository) buildSearchSQL(query string, itemTypes []string, limit, argIdx := 2 // All title-side text on both sides of @@ flows through - // public.normalize_search_text() (migration 127), which strips - // non-alphanumeric chars and standalone "and" tokens. This makes "&" and - // "and" interchangeable so "Law & Order" and "Law and Order" match the - // same items. The expression must match idx_media_items_search_title_fields - // exactly for the GIN index to be used. + // public.normalize_search_text() (migrations 127 / 138), which strips + // non-alphanumeric chars, drops standalone "and" tokens, and normalizes + // common title numbers. The expression must match + // idx_media_items_search_title_fields exactly for the GIN index to be used. // // Overview uses the 'english' config which natively treats "and" as a // stop word, so it does not need explicit normalization. diff --git a/internal/catalog/item_repo_test.go b/internal/catalog/item_repo_test.go index c53d7fc2..d39f2784 100644 --- a/internal/catalog/item_repo_test.go +++ b/internal/catalog/item_repo_test.go @@ -238,7 +238,7 @@ func TestItemRepo_Search_StrictTitleFilter_UsesWindowCount(t *testing.T) { // // The original_title and sort_title fallbacks are intentionally not stored // as generated columns (less search traffic), so they call the -// public.normalize_search_text() function (migration 127) inline. +// public.normalize_search_text() function (migrations 127 / 138) inline. func TestItemRepo_Search_UsesTitleNormalizedColumn(t *testing.T) { repo := &ItemRepository{} sql, _, _ := repo.buildSearchSQL("avatar", []string{"movie"}, 20, 0, AccessFilter{}) @@ -260,9 +260,9 @@ func TestItemRepo_Search_UsesTitleNormalizedColumn(t *testing.T) { // text is wrapped in public.normalize_search_text() before being handed to // websearch_to_tsquery on the title arm, and to phraseto_tsquery for the // phrase rank. The tsvector side of @@ applies the same normalization, so -// "&" and "and" become interchangeable end-to-end (migration 127). The -// overview arm is intentionally left unwrapped — the 'english' config -// already treats "and" as a stop word. +// title normalization stays symmetric end-to-end. The overview arm is +// intentionally left unwrapped — the 'english' config already treats "and" +// as a stop word. func TestItemRepo_Search_NormalizesTsqueryInput(t *testing.T) { repo := &ItemRepository{} sql, _, _ := repo.buildSearchSQL("law and order", []string{"movie"}, 20, 0, AccessFilter{}) diff --git a/internal/catalog/search_query.go b/internal/catalog/search_query.go index ebac4c7a..8d9098ef 100644 --- a/internal/catalog/search_query.go +++ b/internal/catalog/search_query.go @@ -99,7 +99,7 @@ func parseYearToken(token string) (int, bool) { } // normalizeTitleForComparison must stay in lockstep with the SQL function -// public.normalize_search_text (migration 127) and the title_normalized +// public.normalize_search_text (migrations 127 / 138) and the title_normalized // generated column. Mismatches between Go and SQL normalization produce // asymmetric search results (Go-computed ExactTitleHint failing to match a // row whose title_normalized has the same logical content). @@ -116,14 +116,14 @@ func normalizeTitleForComparison(input string) string { } } - return stripStandaloneAndTokens(collapseSearchWhitespace(b.String())) + return normalizeSearchTokens(collapseSearchWhitespace(b.String())) } -// stripStandaloneAndTokens drops the standalone token "and" from a -// whitespace-separated lowercase string. Together with the alphanumeric -// pass above, this makes "&" and the word "and" interchangeable: both -// "Law & Order" and "Law and Order" reduce to "law order". -func stripStandaloneAndTokens(input string) string { +// normalizeSearchTokens drops the standalone token "and" from a +// whitespace-separated lowercase string and maps common number words / +// ordinals to digit tokens. Together with the alphanumeric pass above, this +// mirrors public.normalize_search_text(). +func normalizeSearchTokens(input string) string { if input == "" { return "" } @@ -133,11 +133,85 @@ func stripStandaloneAndTokens(input string) string { if f == "and" { continue } - filtered = append(filtered, f) + filtered = append(filtered, normalizeSearchNumberToken(f)) } return strings.Join(filtered, " ") } +func normalizeSearchNumberToken(token string) string { + switch token { + case "zero", "zeroth": + return "0" + case "one", "first": + return "1" + case "two", "second": + return "2" + case "three", "third": + return "3" + case "four", "fourth": + return "4" + case "five", "fifth": + return "5" + case "six", "sixth": + return "6" + case "seven", "seventh": + return "7" + case "eight", "eighth": + return "8" + case "nine", "ninth": + return "9" + case "ten", "tenth": + return "10" + case "eleven", "eleventh": + return "11" + case "twelve", "twelfth": + return "12" + case "thirteen", "thirteenth": + return "13" + case "fourteen", "fourteenth": + return "14" + case "fifteen", "fifteenth": + return "15" + case "sixteen", "sixteenth": + return "16" + case "seventeen", "seventeenth": + return "17" + case "eighteen", "eighteenth": + return "18" + case "nineteen", "nineteenth": + return "19" + case "twenty", "twentieth": + return "20" + } + + if stripped, ok := stripDigitOrdinalSuffix(token); ok { + return stripped + } + return token +} + +func stripDigitOrdinalSuffix(token string) (string, bool) { + for _, suffix := range []string{"st", "nd", "rd", "th"} { + stem := strings.TrimSuffix(token, suffix) + if stem != token && hasOnlyDigits(stem) { + return stem, true + } + } + return "", false +} + +func hasOnlyDigits(input string) bool { + if input == "" { + return false + } + for _, r := range input { + if !unicode.IsDigit(r) { + return false + } + } + return true +} + func collapseSearchWhitespace(input string) string { return strings.Join(strings.Fields(input), " ") } diff --git a/internal/catalog/search_query_test.go b/internal/catalog/search_query_test.go index b92e4e57..0a2a7477 100644 --- a/internal/catalog/search_query_test.go +++ b/internal/catalog/search_query_test.go @@ -2,13 +2,10 @@ package catalog import "testing" -// TestNormalizeTitleForComparison_AmpersandAndEquivalence pins the contract -// that "&" and the word "and" are interchangeable. This mirrors the SQL -// function public.normalize_search_text() (migration 127). If the Go-side -// normalization drifts from the SQL one, ExactTitleHint will fail to match -// the title_normalized generated column for items whose stored title used -// the other form. -func TestNormalizeTitleForComparison_AmpersandAndEquivalence(t *testing.T) { +// TestNormalizeTitleForComparison pins the Go mirror of the SQL function +// public.normalize_search_text(). If Go-side normalization drifts from SQL, +// ExactTitleHint will fail to match the title_normalized generated column. +func TestNormalizeTitleForComparison(t *testing.T) { cases := []struct { name string in string @@ -25,6 +22,15 @@ func TestNormalizeTitleForComparison_AmpersandAndEquivalence(t *testing.T) { {"consecutive and tokens", "rock and and roll", "rock roll"}, {"acronym with ampersand", "S&P 500", "s p 500"}, {"unicode letters", "Café & Crème", "café crème"}, + {"number word", "Dune: Part Two", "dune part 2"}, + {"number digit", "Dune Part 2", "dune part 2"}, + {"ordinal word", "Second Act", "2 act"}, + {"ordinal digit suffix", "2nd Act", "2 act"}, + {"thirteenth word", "Friday the Thirteenth", "friday the 13"}, + {"thirteenth digit suffix", "Friday the 13th", "friday the 13"}, + {"twenty word", "Twenty", "20"}, + {"composed numbers not folded", "Twenty One", "20 1"}, + {"embedded digit word unchanged", "Se7en", "se7en"}, {"empty stays empty", "", ""}, {"whitespace only", " ", ""}, } @@ -39,10 +45,10 @@ func TestNormalizeTitleForComparison_AmpersandAndEquivalence(t *testing.T) { } } -// TestParseSearchQuery_ExactTitleHint_StripsAnd ensures that the higher-level -// parser propagates the and-stripping normalization into ExactTitleHint, so +// TestParseSearchQuery_ExactTitleHint_NormalizesSearchText ensures that the +// higher-level parser propagates search normalization into ExactTitleHint, so // callers building SQL with the hint get the form that matches title_normalized. -func TestParseSearchQuery_ExactTitleHint_StripsAnd(t *testing.T) { +func TestParseSearchQuery_ExactTitleHint_NormalizesSearchText(t *testing.T) { cases := []struct { in string want string @@ -50,6 +56,8 @@ func TestParseSearchQuery_ExactTitleHint_StripsAnd(t *testing.T) { {"Law & Order", "law order"}, {"Law and Order", "law order"}, {"\"Law and Order\" 2010", "law order"}, + {"Dune: Part Two", "dune part 2"}, + {"\"Dune: Part Two\" 2024", "dune part 2"}, } for _, tc := range cases { t.Run(tc.in, func(t *testing.T) { diff --git a/migrations/138_search_number_word_normalization.down.sql b/migrations/138_search_number_word_normalization.down.sql new file mode 100644 index 00000000..0a841bef --- /dev/null +++ b/migrations/138_search_number_word_normalization.down.sql @@ -0,0 +1,38 @@ +-- Restore the pre-138 normalization: collapse non-alnum to spaces, lowercase, +-- and strip standalone "and" tokens, but do not normalize number words or +-- ordinal digit suffixes. + +DROP INDEX IF EXISTS public.idx_media_items_search_title_fields; +DROP INDEX IF EXISTS public.idx_media_items_title_normalized_trgm; + +ALTER TABLE public.media_items DROP COLUMN IF EXISTS title_normalized; + +CREATE OR REPLACE FUNCTION public.normalize_search_text(input text) +RETURNS text +LANGUAGE sql +IMMUTABLE +PARALLEL SAFE +AS $$ + SELECT BTRIM(REGEXP_REPLACE( + ' ' || BTRIM(LOWER(REGEXP_REPLACE(COALESCE(input, ''), '[^[:alnum:]]+', ' ', 'g'))) || ' ', + ' (and )+', + ' ', + 'g' + )); +$$; + +ALTER TABLE public.media_items + ADD COLUMN title_normalized text + GENERATED ALWAYS AS (public.normalize_search_text(title)) STORED; + +CREATE INDEX IF NOT EXISTS idx_media_items_title_normalized_trgm + ON public.media_items USING gin (title_normalized public.gin_trgm_ops); + +CREATE INDEX IF NOT EXISTS idx_media_items_search_title_fields + ON public.media_items USING gin (( + setweight(to_tsvector('simple', public.normalize_search_text(COALESCE(title, ''))), 'A') || + setweight(to_tsvector('simple', public.normalize_search_text(COALESCE(original_title, ''))), 'A') || + setweight(to_tsvector('simple', public.normalize_search_text(COALESCE(sort_title, ''))), 'B') + )); + +DROP FUNCTION IF EXISTS public.normalize_search_number_token(text); diff --git a/migrations/138_search_number_word_normalization.up.sql b/migrations/138_search_number_word_normalization.up.sql new file mode 100644 index 00000000..902170a6 --- /dev/null +++ b/migrations/138_search_number_word_normalization.up.sql @@ -0,0 +1,116 @@ +-- OPERATOR NOTE: This migration drops and re-adds the title_normalized +-- STORED generated column on media_items, which rewrites the entire table +-- under an ACCESS EXCLUSIVE lock. It also rebuilds the title search GIN +-- indexes, so search queries will fall back to slower plans while the +-- indexes are being rebuilt. Run during a maintenance window on large +-- libraries. +-- +-- Goal: make common title numbers interchangeable between word and digit +-- forms in FTS search, so "Dune: Part Two" and "Dune Part 2" match the same +-- title-side tokens. + +DROP INDEX IF EXISTS public.idx_media_items_search_title_fields; +DROP INDEX IF EXISTS public.idx_media_items_title_normalized_trgm; + +ALTER TABLE public.media_items DROP COLUMN IF EXISTS title_normalized; + +CREATE OR REPLACE FUNCTION public.normalize_search_number_token(token text) +RETURNS text +LANGUAGE sql +IMMUTABLE +PARALLEL SAFE +AS $$ + WITH normalized AS ( + SELECT LOWER(COALESCE(token, '')) AS value + ) + SELECT CASE value + WHEN 'zero' THEN '0' + WHEN 'zeroth' THEN '0' + WHEN 'one' THEN '1' + WHEN 'first' THEN '1' + WHEN 'two' THEN '2' + WHEN 'second' THEN '2' + WHEN 'three' THEN '3' + WHEN 'third' THEN '3' + WHEN 'four' THEN '4' + WHEN 'fourth' THEN '4' + WHEN 'five' THEN '5' + WHEN 'fifth' THEN '5' + WHEN 'six' THEN '6' + WHEN 'sixth' THEN '6' + WHEN 'seven' THEN '7' + WHEN 'seventh' THEN '7' + WHEN 'eight' THEN '8' + WHEN 'eighth' THEN '8' + WHEN 'nine' THEN '9' + WHEN 'ninth' THEN '9' + WHEN 'ten' THEN '10' + WHEN 'tenth' THEN '10' + WHEN 'eleven' THEN '11' + WHEN 'eleventh' THEN '11' + WHEN 'twelve' THEN '12' + WHEN 'twelfth' THEN '12' + WHEN 'thirteen' THEN '13' + WHEN 'thirteenth' THEN '13' + WHEN 'fourteen' THEN '14' + WHEN 'fourteenth' THEN '14' + WHEN 'fifteen' THEN '15' + WHEN 'fifteenth' THEN '15' + WHEN 'sixteen' THEN '16' + WHEN 'sixteenth' THEN '16' + WHEN 'seventeen' THEN '17' + WHEN 'seventeenth' THEN '17' + WHEN 'eighteen' THEN '18' + WHEN 'eighteenth' THEN '18' + WHEN 'nineteen' THEN '19' + WHEN 'nineteenth' THEN '19' + WHEN 'twenty' THEN '20' + WHEN 'twentieth' THEN '20' + ELSE + CASE + WHEN value ~ '^[0-9]+(st|nd|rd|th)$' THEN REGEXP_REPLACE(value, '(st|nd|rd|th)$', '') + ELSE value + END + END + FROM normalized; +$$; + +-- Single source of truth for search text normalization. Used by: +-- * media_items.title_normalized generated column +-- * idx_media_items_search_title_fields GIN expression +-- * Inline original_title / sort_title normalization in buildSearchSQL +-- * websearch_to_tsquery() argument wrapping +-- +-- Strips non-alphanumeric chars to spaces, lowercases, drops standalone +-- "and" tokens, normalizes common number words / ordinals to digit tokens, +-- and rejoins tokens in original order. Returns '' when input is NULL. +CREATE OR REPLACE FUNCTION public.normalize_search_text(input text) +RETURNS text +LANGUAGE sql +IMMUTABLE +PARALLEL SAFE +AS $$ + SELECT COALESCE( + STRING_AGG(public.normalize_search_number_token(token), ' ' ORDER BY ord), + '' + ) + FROM REGEXP_SPLIT_TO_TABLE( + BTRIM(LOWER(REGEXP_REPLACE(COALESCE(input, ''), '[^[:alnum:]]+', ' ', 'g'))), + '[[:space:]]+' + ) WITH ORDINALITY AS tokens(token, ord) + WHERE token <> '' AND token <> 'and'; +$$; + +ALTER TABLE public.media_items + ADD COLUMN title_normalized text + GENERATED ALWAYS AS (public.normalize_search_text(title)) STORED; + +CREATE INDEX IF NOT EXISTS idx_media_items_title_normalized_trgm + ON public.media_items USING gin (title_normalized public.gin_trgm_ops); + +CREATE INDEX IF NOT EXISTS idx_media_items_search_title_fields + ON public.media_items USING gin (( + setweight(to_tsvector('simple', public.normalize_search_text(COALESCE(title, ''))), 'A') || + setweight(to_tsvector('simple', public.normalize_search_text(COALESCE(original_title, ''))), 'A') || + setweight(to_tsvector('simple', public.normalize_search_text(COALESCE(sort_title, ''))), 'B') + ));