Files
silo-server/internal/recommendations/embeddings/text.go
T
RXWatcherandClaude Opus 4.7 14a05bba9f feat(audiobooks): comprehensive UX, scanner, and collections work
Detail page:
- collapse Chapters section behind header toggle (73-chapter pages no
  longer push content below the fold)
- square cover frames throughout (audiobook covers are Audible-style 1:1,
  not 2:3 book portrait)
- narrator picker dropdown when multiple narrations of the same book exist
- clickable genre badges (route to /audiobooks?genre=X)
- reordered so credits/rails sit above the chapter list
- Play-from-Start button forces remount via playToken counter (was a
  no-op when player was already at position 0)
- regression test for the Play-from-Start fix

Mini bar / Now Listening:
- mini bar respects --app-sidebar-offset so it stops getting covered by
  the desktop sidebar
- Now Listening adds overflow scroll + a labeled "Back to player" button
  so controls are never inaccessible on short viewports

Library page:
- infinite scroll (replaces Previous/Next pagination)
- genre filter chip with X-to-clear

Scanner:
- 8-worker parallel reconcile (env SILO_AUDIOBOOK_SCAN_WORKERS to override)
- file-path-first dedup so cleaned titles don't collapse separate
  narrations
- title cleanup at write time (strips "Read by X" / "(unabridged)"
  suffixes; original tag preserved in original_title)
- audiobook_series upsert from tag-derived series_name/series_position
- secondary dedup check (author + narrator + year + duration ±0.5% +
  title-prefix) so two folders of the same book attach to one row

Backend detail handler:
- new fetchAlsoByAuthor, fetchInSeries (with series row > 1 entry guard),
  fetchSimilar (embedding-first with shared-genre fallback),
  fetchOtherNarrations (regex-strips narrator suffix to group siblings)
- audiobookDetailResponse gained also_by_author, in_series,
  similar_audiobooks, other_narrations fields
- list endpoint accepts a genre query param

Embeddings:
- BuildEmbeddingText branches on item.Type == "audiobook" to use
  author/narrator credits instead of cast/director/writer
- mediaTypeLabel helper centralizes movie / "TV series" / audiobook
- ListEmbeddingTextCandidates SQL mirrors the Go branching exactly
- ItemsNeedingEmbedding and TotalMediaItemCount loosen status='matched'
  gate to also include audiobooks (which don't go through TMDB match)
- FindSimilar gains a mediaType filter so cross-type results never appear
- callers in similar.go and personal.go pass the source item's type

Collections:
- MediaAudiobook MediaKind + audiobook(s) case in templateEligibleForLibrary
  (stops offering broken movie/TV templates to audiobook libraries)
- useAddItemToCollection hook (user + admin/library endpoints)
- AddToCollectionDialog wired into audiobook detail and movie/series
  ActionBar overflow menu
- ManualCollectionItemsEditor gained a search-and-add panel with
  debounced live results
- QueryDefinition.media_scope, QuerySortRelevanceScope, ALL_MEDIA_SCOPES
  extended to include "audiobook"
- CatalogFilterBar gained an Audiobooks media scope option
- parseCatalogMediaScope (backend) accepts "audiobook"

Migrations:
- 145_audiobook_series: per-book series_name/series_index with a
  best-effort title-pattern backfill for the existing corpus
- 146_audiobook_title_cleanup: strips narrator suffix / (unabridged)
  noise from existing titles, preserving raw in original_title

Scripts:
- scripts/dedup_audiobooks.py: one-shot merge for "Title" vs
  "Title: Subtitle" duplicates, file-path-stable, dry-run by default

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-25 19:47:54 +02:00

204 lines
5.5 KiB
Go

package embeddings
import (
"fmt"
"sort"
"strings"
"github.com/Silo-Server/silo-server/internal/models"
)
const (
maxOverviewRunes = 1000
maxKeywords = 5
)
func truncateRunes(s string, limit int) string {
if limit <= 0 {
return ""
}
count := 0
for i := range s {
if count == limit {
return s[:i]
}
count++
}
return s
}
// BuildEmbeddingText constructs a structured text representation of a media item
// suitable for generating embeddings. Leads with semantic content (genres + overview)
// to prevent title-word dominance in the embedding space.
func BuildEmbeddingText(item *models.MediaItem) string {
var parts []string
// Lead with genres + type + overview for semantic dominance.
typeName := mediaTypeLabel(item.Type)
if len(item.Genres) > 0 && item.Overview != "" {
overview := truncateRunes(item.Overview, maxOverviewRunes)
parts = append(parts, fmt.Sprintf("%s %s about %s", strings.Join(item.Genres, ", "), typeName, overview))
} else if len(item.Genres) > 0 {
parts = append(parts, fmt.Sprintf("%s %s", strings.Join(item.Genres, ", "), typeName))
} else if item.Overview != "" {
overview := truncateRunes(item.Overview, maxOverviewRunes)
parts = append(parts, fmt.Sprintf("%s. %s", typeName, overview))
}
// Title with year — present but no longer leading.
if item.Year > 0 {
parts = append(parts, fmt.Sprintf("%s (%d)", item.Title, item.Year))
} else {
parts = append(parts, item.Title)
}
if item.ContentRating != "" {
parts = append(parts, fmt.Sprintf("Rated %s", item.ContentRating))
}
if item.Tagline != "" {
parts = append(parts, fmt.Sprintf(`"%s"`, item.Tagline))
}
if item.Type == "audiobook" {
// Audiobooks credit author (kind=7) and narrator (kind=8). Cast/
// director/writer don't apply. Keep the SQL canonical_text builder
// in recommendations/repo.go in sync with this shape.
var authors, narrators []models.ItemPerson
for _, p := range item.People {
switch p.Kind {
case models.PersonKindAuthor:
authors = append(authors, p)
case models.PersonKindNarrator:
narrators = append(narrators, p)
}
}
sortItemPeople(authors)
sortItemPeople(narrators)
if len(authors) > 0 {
parts = append(parts, fmt.Sprintf("Written by %s", strings.Join(itemPersonNames(authors), ", ")))
}
if len(narrators) > 0 {
parts = append(parts, fmt.Sprintf("Narrated by %s", strings.Join(itemPersonNames(narrators), ", ")))
}
} else {
// Top-billed cast with character names (up to 5).
var actors []models.ItemPerson
for _, p := range item.People {
if p.Kind == models.PersonKindActor {
actors = append(actors, p)
}
}
sortItemPeople(actors)
if len(actors) > 0 {
credits := make([]string, 0, 5)
for i, p := range actors {
if i >= 5 {
break
}
if p.Character != "" {
credits = append(credits, fmt.Sprintf("%s as %s", p.Name, p.Character))
} else {
credits = append(credits, p.Name)
}
}
parts = append(parts, fmt.Sprintf("Cast: %s", strings.Join(credits, ", ")))
}
// Director(s).
var directors []models.ItemPerson
for _, p := range item.People {
if p.Kind == models.PersonKindDirector {
directors = append(directors, p)
}
}
sortItemPeople(directors)
if len(directors) > 0 {
parts = append(parts, fmt.Sprintf("Directed by %s", strings.Join(itemPersonNames(directors), ", ")))
}
// Writer(s).
var writers []models.ItemPerson
for _, p := range item.People {
if p.Kind == models.PersonKindWriter {
writers = append(writers, p)
}
}
sortItemPeople(writers)
if len(writers) > 0 {
parts = append(parts, fmt.Sprintf("Written by %s", strings.Join(itemPersonNames(writers), ", ")))
}
}
if len(item.Keywords) > 0 {
keywords := item.Keywords
if len(keywords) > maxKeywords {
keywords = keywords[:maxKeywords]
}
parts = append(parts, fmt.Sprintf("Keywords: %s", strings.Join(keywords, ", ")))
}
if item.OriginalLanguage != "" {
parts = append(parts, fmt.Sprintf("Original language: %s", item.OriginalLanguage))
}
if len(item.Studios) > 0 {
parts = append(parts, fmt.Sprintf("Studios: %s", strings.Join(item.Studios, ", ")))
}
if len(item.Networks) > 0 {
parts = append(parts, fmt.Sprintf("Network: %s", strings.Join(item.Networks, ", ")))
}
if len(item.Countries) > 0 {
countries := item.Countries
if len(countries) > 2 {
countries = countries[:2]
}
parts = append(parts, fmt.Sprintf("Country: %s", strings.Join(countries, ", ")))
}
return strings.ToValidUTF8(strings.Join(parts, ". "), "")
}
// mediaTypeLabel maps a media_items.type to the natural-language label
// used in the embedding canonical text. The exact strings here are part
// of the embedding-text contract — change them and every existing
// embedding goes stale (see canonicalText logic in repo.go).
func mediaTypeLabel(t string) string {
switch t {
case "series":
return "TV series"
case "audiobook":
return "audiobook"
default:
return "movie"
}
}
func sortItemPeople(people []models.ItemPerson) {
sort.SliceStable(people, func(i, j int) bool {
if people[i].SortOrder != people[j].SortOrder {
return people[i].SortOrder < people[j].SortOrder
}
if people[i].Name != people[j].Name {
return people[i].Name < people[j].Name
}
if people[i].Character != people[j].Character {
return people[i].Character < people[j].Character
}
return people[i].ID < people[j].ID
})
}
func itemPersonNames(people []models.ItemPerson) []string {
names := make([]string, 0, len(people))
for _, p := range people {
names = append(names, p.Name)
}
return names
}