Files
silo-server/internal/recommendations/personal.go
8b70357703 feat(ebooks): first-class ebook libraries, scanner, and reader (#124)
* docs: define ebook architecture matching audiobooks

* docs: plan ebook audiobook-parity implementation

* feat: add ebook scanner parser foundation

* fix: harden ebook scanner foundation

* fix: handle ebook isbn labels

* fix: guard ebook subtree scans

* feat: scan ebook libraries in core

* fix: preserve ebook scan people credits

* fix: refresh ebook scan metadata safely

* feat: persist ebook series membership

* test: cover ebook series persistence decisions

* fix: address ebook scanner PR review

* docs: clarify ebook foundation PR scope

* feat: add ebook metadata enricher

* fix: harden ebook poster cache

* feat: wire ebook metadata sync task

* feat: expose ebook library metadata setup

* feat: add ebook catalog scope support

* feat: add ebook detail view

* feat: label ebook file versions by format

* feat: use file-size copy for downloads

* feat: use file language in download dialog

* test: cover ebook detail authors and downloads

* fix: drop narrator credits from ebook scanner merges

* fix: align ebook collection filters with book media

* fix: drop asin provider ids from ebook enrichment

* fix: force ebook people refresh for stale narrators

* chore: omit ebook planning docs from branch

* feat: add ebook detail related content

* feat: add ebook reader file entrypoint

* feat: render ebooks with foliate reader

* feat: persist ebook reader progress

* feat: add ebook reader controls

* feat: extract ebook pdf metadata

* feat: favor scanner isbn during ebook enrichment

* feat: extract fbz ebook metadata

* feat: count cbz ebook pages

* feat: show ebook file page counts

* feat: show ebook download summaries

* feat: switch ebook reader files

* feat: prefer epub for ebook read action

* feat: surface ebook reader progress

* feat: sync ebook reader progress cache

* feat: hide ebook read action for unsupported files

* feat: filter ebook reader file selector

* fix: serve fbz ebook archives with reader mime type

* fix: detect fbz ebooks from compound filename

* fix: authorize fbz ebooks from compound filename

* fix: scope ebook catalog facets

* fix: reject narrator queries for ebooks

* fix: build ebook recommendation text from authors

* fix: include ebooks in embedding eligibility

* fix: include ebooks in recommendation media mix

* fix: include ebooks in recently added recommendations

* feat: include ebook progress in recommendation signals

* feat: include ebooks in continue watching sections

* feat: include ebooks in catalog progress metrics

* fix: read ebook isbn from epub metadata

* fix: filter ebook asin provider aliases

* fix: fall back from unsupported ebook reader files

* fix: sort ebook catalogs by reader progress

* fix: filter ebook catalogs by reader progress

* fix: include ebooks in last watched catalog filters

* feat: reflect ebook reader progress in item user state

* feat: share ebook progress state across item surfaces

* feat: report ebook scan progress

* fix: include ebook activity in recommendations

* fix: expose ebook reader progress on item detail

* fix: support ebook subtree scans

* fix: honor profile header for ebook item progress

* fix: add ebook library default sections

* fix: route ebook continue cards to reader

* fix: hide watched toggle for ebooks

* fix: route ebook watch tonight cards to reader

* fix: route ebook hero actions to reader

* fix: detect archive ebook reader formats by filename

* feat: cache embedded ebook covers during scan

* fix: encode ebook hero reader links

* fix: persist non-epub ebook reader progress

* fix: scope narrator catalog badges to audiobooks

* fix: merge ebook reader progress during item repair

* fix: label ebook progress filters as read

* fix: show ebook related rails as book covers

* fix: remove txt ebook reader support

* fix: reject txt ebook reader files

* fix: label ebook advanced filters as read

* fix: label ebook personalized sorts as read

* fix: remove plain text reader loader path

* test: cover ebook unread catalog rules

* fix: preserve ebook reader library context

* fix: link ebook genres with library scope

* fix: encode related rail item links

* fix: encode catalog card item links

* fix: encode hero and continue item links

* fix: encode watch tonight item links

* fix: encode recommendation and search item links

* test: cover ebook scan format set

* fix: label ebook search results clearly

* fix: make global search prompt media neutral

* fix: encode catalog read API ids

* fix: encode item API ids

* fix: include ebook reader vendor in docker build

* fix: make ebook reader build clean

* fix: clean ebook embedded descriptions

* docs: plan ebook reader shell parity

* feat: add ebook reader shell controls

* fix: widen ebook scrolled reader flow

* fix: remove scrolled reader content width cap

* docs: plan ebook reader full parity

* feat: persist ebook reader config

* feat: add ebook annotations and bookmarks

* feat: add ebook reader tools and aids

* feat: add ebook advanced reader settings

* fix: keep ebook reader panel in viewport

* fix: use foliate sizing units for ebook scroll flow

* fix: keep ebook settings controls readable

* fix: simplify ebook reader settings controls

* feat(ebooks): extract local covers during scan (#98)

* feat(ebooks): extract local covers during scan

* fix(ebooks): read nullable poster paths during cover scan

* fix(catalog): coalesce nullable media artwork fields

* fix(ebooks): group sibling formats by book identity

* fix(ebooks): tolerate legacy ebook metadata encodings

* fix(ebooks): decode PDF hex metadata strings

* fix(ebooks): harden local cover extraction and format grouping

Address review findings on the local cover scan:

- Restrict generic sidecar covers (cover.jpg, folder.png, ...) to
  single-book directories, always accept images named after the book
  file, and apply exactly one cover per reconcile with sidecar taking
  precedence over the embedded cover.
- Replace the read-then-write poster update with an atomic conditional
  UPDATE (ItemRepository.SetLocalPoster) so provider/admin artwork is
  never clobbered by concurrent writers, and refresh locally owned
  posters when the extracted cover bytes change (thumbhash compare).
- Preserve UTF-8 PDF Info strings (including a UTF-8 BOM) instead of
  forcing everything through Windows-1252; the cp1252 fallback now only
  applies to non-UTF-8 bytes.
- Select EPUB covers by manifest media-type with properties="cover-image"
  outranking the EPUB2 meta name="cover" id, so XHTML cover pages no
  longer shadow the real image.
- Order CBZ pages naturally (2.jpg before 10.jpg, ch2/ before ch10/)
  when picking the cover page, via a single O(n) min-scan.
- Bump the ebook content group key scheme to version 2 and reprocess
  rows written under older versions so pre-existing libraries gain
  sibling-format grouping instead of accumulating duplicates.
- Group different formats only (a same-format sibling with colliding
  sparse metadata stays a separate item) and stop a joining sibling's
  embedded metadata from overwriting a provider-matched item.
- Decode any IANA-labelled OPF/FB2 XML charset (windows-1251, koi8-r,
  shift_jis, ...) via x/net/html/charset, and wire the charset reader
  into FB2 parsing which previously had none.
- Strip the full .fb2.zip double extension from filename-derived titles
  and group keys.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

---------

Co-authored-by: rxwatcher <rxwatcher@users.noreply.github.com>
Co-authored-by: Quick <31828688+Quick104@users.noreply.github.com>
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>

* feat(ebooks): add reader profiles and ruler (#99)

* feat(ebooks): extract local covers during scan

* fix(ebooks): read nullable poster paths during cover scan

* fix(catalog): coalesce nullable media artwork fields

* fix(ebooks): group sibling formats by book identity

* fix(ebooks): tolerate legacy ebook metadata encodings

* fix(ebooks): decode PDF hex metadata strings

* feat(ebooks): add reader profiles and ruler

* fix(ebooks): address reader ruler and profile review findings

- skip renderer setStyles/render when computed styles and attributes are
  unchanged, so ruler position updates no longer re-style the book view
- drag the ruler via a local draft that commits on release, with the
  surface rect cached at pointer-down
- migrate font values persisted before the generic stacks (Inter,
  Georgia, Merriweather, legacy serif) so the font select never renders
  blank, with a Custom fallback option for unknown values
- make the ruler band click-through and move dragging to a dedicated
  keyboard-accessible slider handle so links and text selection keep
  working under the band
- share font stacks between options and profiles via READER_FONT_STACKS
- surface the active reading profile, move presets to the top of the
  settings panel, and drop the redundant profile button aria-labels

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(ebooks): resolve prefer-const lint error in readest document lib

`pnpm run lint` failed on the branch because `direction` is never
reassigned in getDirection; split the destructure so only the
reassigned `writingMode` stays mutable.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

---------

Co-authored-by: rxwatcher <rxwatcher@users.noreply.github.com>
Co-authored-by: Quick <31828688+Quick104@users.noreply.github.com>
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>

* Merge branch 'main' into work/ebooks-reader-base

Brings the ebook integration branch up to date with main (audiobook
library redesign, continue-watching rework and card affordances,
quic-go bump, jellycompat fixes). Conflict resolutions favor main's
generalized mechanisms and register ebooks with them:

- media scope validation goes through IsValidMediaScope (now including
  "ebook" alongside main's "video" group scope), in Go and in the web
  filter/search types
- continue-watching uses main's typed rails; reading-type sections pull
  resume points from ebook_reader_progress and the ebook library default
  section is wired to ContinueTypeConfig(ContinueTypeReading)
- item_repo keeps main's derived select-list machinery (itemColumnExpr)
  and both poster accessors (GetPoster/SetLocalPoster for ebook covers,
  GetPosterPath for audiobook covers)
- web cards/hero/watch-tonight adopt main's buildMediaPlayHref helpers,
  which now route ebooks to /reader/ebook and encode content ids;
  ebook affordances (BookOpen icon, Read verb, percent-read subtitle)
  carry over onto main's reworked components
- LibraryForm ebook support ported into main's refactored
  useLibraryForm/libraryTypes modules

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(docker): copy foliate-js vendor into Dockerfile.dev frontend stage

foliate-js is a file:vendor/foliate-js dependency, so pnpm install needs
the vendor directory before the lockfile install layer. The production
Dockerfile already copies it; the dev image was missed, breaking
make dev-deploy with ENOENT on /app/web/vendor/foliate-js.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* feat(ebooks): render Continue Reading sections as upright poster cards

All-ebook continue sections previously fell through to the horizontal
16:9 wide card; include ebooks in the poster-variant check so book
covers render in their natural 2:3 framing.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(ui): stop related-rail highlight ring clipping on detail pages

Move the current-item ring onto the cover artwork with a themed
ring-offset color (matching the sidebar profile highlight) and give
the scroll container top headroom so the ring is not cut off by
overflow-x-auto. Applies to both ebook and audiobook detail rails.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(scanner): harden ebook scanning against data loss and bad metadata

- Reconcile missing ebook files like video/audio, with real per-root walk
  failure tracking (failed/unmounted roots are excluded from deletion),
  symlinked-root support via the shared logical walker, and the empty-root
  cleanup allowance before any destructive reconciliation.
- Create ebook items as 'pending' so enrichment can promote them to
  'matched' (backfill migration included), and protect matched items from
  re-scan clobbering: title/year skipped, people/series fill-empty only.
- PDF metadata: scan head + tail windows (non-linearized PDFs keep the Info
  dict at the end), require proper key delimiters, head values win.
- Cap plain .fb2 reads like .fbz entries; drop .md as an ebook format.
- gofmt internal/scanner/audiobook.go (pre-existing drift).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(ebooks): make enrichment failures non-terminal with dedicated backoff state

- Provider errors now record a failure (capped retries) instead of stamping
  last_refreshed, which permanently excluded items after transient outages.
- Unconfigured metadata chains and the scan-window membership race skip the
  item without stamping or burning a retry.
- Failure tracking moves to a new ebook_enrichment_state table, decoupling
  it from media_items.refresh_failures (shared with metadata refresh debt).
- Preserve non-author people credits when persisting enrichment results.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(catalog): gate ebook progress on hidden history and centralize threshold

- Apply user_history_hidden_items gating (video semantics) to the ebook
  watched/in-progress filters, progress sort plan, and Continue Reading.
- Continue Reading pages past dismissed items via the shared collector and
  dedupes items across pages (also fixes the video path's latent exposure).
- Centralize the 0.9 finished threshold as models.EbookFinishedProgressThreshold
  with a single SQL-interpolated mirror in catalog.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(recommendations): correct watcher counting and wire ebook taste signals

- itemWatchersQuery dedupes to distinct (watcher, item) rows so one
  binge-watcher can no longer satisfy minWatchers; the eligibility floor
  now counts distinct accounts rather than profiles.
- Hidden-history gating on GetEbookReaderProgressForUser (signal reader).
- Ebook reading produces canonical implicit taste signals (weighted like
  the equivalent movie progress ratio); ebooks join taste-seed candidates.
- Stale GetRecentlyAddedItems doc comment corrected.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(api): harden ebook reader endpoints and serve a Content-Security-Policy

- Serve a CSP on all SPA HTML responses: blob/srcdoc book iframes inherit
  it, so script-src 'self' 'wasm-unsafe-eval' blocks script execution from
  malicious book content (sandbox alone is defeated by the WebKit
  allow-scripts requirement). Threat model documented on the constant.
- X-Content-Type-Options: nosniff on frontend, jellycompat, and ebook file
  responses; MIME resolution can no longer fall through to octet-stream
  for an admitted ebook file.
- Annotation PATCH: presence-aware field semantics (absent keeps, present
  sets/clears), invariant re-validation on the merged row, and an atomic
  SELECT ... FOR UPDATE read-merge-write.
- Request size caps (413) on progress/config/annotation writes;
  Content-Disposition via mime.FormatMediaType; hidden-history gating in
  the shared ebook progress lister; FK-cascade indexes for reader tables.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* feat(api): native read-state endpoints for ebooks

- POST/DELETE /watched/{id} accepts ebook content IDs: mark read upserts
  progress 1.0 preserving the reader's file/location (or picks the
  preferred reader file for never-opened books); mark unread mirrors video
  unwatch semantics and deletes the progress row.
- /history/remove accepts ebooks: hides via user_history_hidden_items
  without touching the reading position (hidden != unread; next reading
  activity resurfaces the book, mirroring video re-watch).
- Access-filter checks match the video branch; shared logic lives in
  ebook_read_state.go. Sort metrics/user-state thresholds use the shared
  constant; profile-header fallback deduplicated.

Clients: response is {type: "ebook", affected_count: 1, played: bool};
the existing watched SSE event fires.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* fix(web): harden the ebook reader UI

- Open-flow race: cancellation checked after every await with full stale-run
  teardown (no wrong-file progress saves, no leaked views/blob URLs);
  book.destroy() on cleanup.
- Progress: monotonic stale-response guard; visibilitychange flush uses the
  refresh-capable client, pagehide uses keepalive; per-book cross-format
  progress documented as deliberate.
- Settings: side effects out of the setState updater; local edits no longer
  clobbered by late server config; pending saves flushed on unmount/pagehide.
- TTS: generation token so Stop actually stops (Chromium/Firefox synthetic
  events); Media Session uninstalled on unmount.
- External book links: http(s) only, opened with noopener,noreferrer.
- apiBlob 512 MiB guard with a user-facing error; fraction bookmarks
  navigable; search-result key collisions fixed; dead e-ink code removed;
  getLibrarySortRelevanceScope deduplicated; md format dropped.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* feat(web): mark read/unread affordances for ebooks

- Item detail gets a Mark Read/Unread button; card menus drop the ebook
  gate and share type-aware labels/toasts (also dedupes audiobook wording).
- Watched-state invalidation includes the reader progress query key so the
  Continue button and percent refresh after toggling.
- Continue Reading dismiss copy for ebooks; dismissal path now URL-encodes
  item IDs (ebook content IDs can contain reserved characters).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* docs: record the PR #124 review and hardening pass

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

---------

Co-authored-by: rxwatcher <rxwatcher@users.noreply.github.com>
Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
2026-06-10 08:18:35 -04:00

459 lines
13 KiB
Go

package recommendations
import (
"context"
"fmt"
"sort"
"time"
"github.com/Silo-Server/silo-server/internal/catalog"
)
const aggregateMediaTypeFloorDivisor = 5
var aggregateSupplementMediaTypes = []string{"movie", "series", "audiobook", "ebook"}
// ForYou returns personalised recommendations grouped by taste clusters.
// For cold-start users, non-personalized rows are returned.
func (e *Engine) ForYou(ctx context.Context, userID int, profileID string, limit int) (*ForYouResponse, error) {
// Check signal count to determine cold-start level.
meta, err := e.repo.GetTasteProfileMeta(ctx, userID, profileID)
if err != nil {
return nil, fmt.Errorf("get taste profile meta: %w", err)
}
positiveSignals := 0
if meta != nil {
for k, v := range meta.SignalCounts {
switch k {
case "rated_low", "watch_low":
// negative signals don't count
default:
positiveSignals += v
}
}
}
level := coldStartLevel(positiveSignals)
// Build cold-start rows (always available).
coldStartRows, err := e.buildColdStartRows(ctx)
if err != nil {
return nil, fmt.Errorf("build cold start rows: %w", err)
}
watchedSet, err := e.watchedItemIDSet(ctx, userID, profileID)
if err != nil {
return nil, fmt.Errorf("get watched item IDs: %w", err)
}
watchedIDs := scoredItemIDsFromSet(watchedSet)
coldStartRows = excludeWatchedRows(coldStartRows, watchedSet)
// If no taste profile at all, return cold-start only.
if meta == nil || level == 0 {
return &ForYouResponse{Rows: coldStartRows}, nil
}
// Build personalized rows from taste clusters.
liveFilter := catalog.AccessFilter{UserID: userID, ProfileID: profileID}
personalRows, err := e.buildClusterRows(ctx, userID, profileID, limit, watchedIDs, liveFilter)
if err != nil {
return nil, fmt.Errorf("build cluster rows: %w", err)
}
aggregatedRow, err := e.buildAggregatedRow(ctx, userID, profileID, limit, watchedIDs, liveFilter)
if err != nil {
return nil, fmt.Errorf("build aggregated row: %w", err)
}
personalRows = combinePersonalRows(aggregatedRow, personalRows)
merged := mergePersonalizedAndColdStart(personalRows, coldStartRows, level)
return &ForYouResponse{Rows: merged}, nil
}
func combinePersonalRows(aggregated *ForYouRow, clusterRows []ForYouRow) []ForYouRow {
if aggregated == nil {
return clusterRows
}
rows := make([]ForYouRow, 0, len(clusterRows)+1)
rows = append(rows, *aggregated)
rows = append(rows, clusterRows...)
return rows
}
// buildClusterRows generates per-cluster recommendation rows.
func (e *Engine) buildClusterRows(ctx context.Context, userID int, profileID string, limit int, excludeIDs []string, filter catalog.AccessFilter) ([]ForYouRow, error) {
clusters, err := e.repo.GetTasteClusters(ctx, userID, profileID)
if err != nil {
return nil, fmt.Errorf("get taste clusters: %w", err)
}
if len(clusters) == 0 {
return nil, nil
}
// Calculate total weight across all clusters for proportional allocation.
var totalWeight float64
for _, c := range clusters {
totalWeight += c.TotalWeight
}
var rows []ForYouRow
for _, c := range clusters {
if c.Embedding == nil || len(c.Embedding) == 0 {
continue
}
// Proportional candidate count.
proportion := 1.0 / float64(len(clusters))
if totalWeight > 0 {
proportion = c.TotalWeight / totalWeight
}
clusterLimit := int(float64(limit) * proportion)
if clusterLimit < 3 {
clusterLimit = 3
}
// Fetch after access and genre constraints so filtered-out items do not
// consume the candidate headroom before MMR.
candidates, _, err := e.repo.FindTasteProfileCandidates(ctx, c.Embedding, excludeIDs, c.DominantGenres, clusterLimit*3, filter)
if err != nil {
continue
}
if len(candidates) == 0 {
continue
}
// Apply MMR re-ranking.
candidateIDs := make([]string, len(candidates))
for i, item := range candidates {
candidateIDs[i] = item.MediaItemID
}
embMap, _ := e.repo.GetBatchEmbeddings(ctx, candidateIDs)
reranked := applyMMR(candidates, embMap, e.mmrLambda(LambdaGenreRow), clusterLimit)
// Apply recency boost.
addedDates, _ := e.repo.GetItemAddedDates(ctx, candidateIDs)
reranked = applyRecencyBoost(reranked, addedDates, time.Now())
label := c.Label
if label == "" {
label = "For You"
}
reason := "Because you enjoy " + label
for i := range reranked {
reranked[i].Reason = reason
}
rows = append(rows, ForYouRow{
Type: "cluster",
Label: reason,
ClusterIndex: c.ClusterIdx,
Items: reranked,
})
}
return rows, nil
}
// buildAggregatedRow builds a single "For You" row from the aggregated taste profile.
func (e *Engine) buildAggregatedRow(ctx context.Context, userID int, profileID string, limit int, excludeIDs []string, filter catalog.AccessFilter) (*ForYouRow, error) {
embedding, err := e.repo.GetTasteProfile(ctx, userID, profileID)
if err != nil {
return nil, fmt.Errorf("get taste profile: %w", err)
}
if embedding == nil {
return nil, nil
}
candidates, genreMap, err := e.repo.FindTasteProfileCandidates(ctx, embedding, excludeIDs, nil, limit*3, filter)
if err != nil {
return nil, fmt.Errorf("find similar for aggregated: %w", err)
}
if len(candidates) == 0 {
return nil, nil
}
candidates, genreMap, mediaTypes := e.addAggregateMediaTypeSupplements(ctx, embedding, excludeIDs, filter, candidates, genreMap, limit)
candidateIDs := make([]string, len(candidates))
for i, item := range candidates {
candidateIDs[i] = item.MediaItemID
}
embMap, _ := e.repo.GetBatchEmbeddings(ctx, candidateIDs)
reranked := applyMMR(candidates, embMap, e.mmrLambda(LambdaForYou), limit)
// Apply genre cap on the main For You row.
reranked = applyGenreCap(reranked, genreMap, GenreCapPercent)
reranked = applyMediaTypeFloor(reranked, candidates, mediaTypes)
for i := range reranked {
reranked[i].Reason = "Personalized for you"
}
return &ForYouRow{
Type: "cluster",
Label: "For You",
Items: reranked,
}, nil
}
func (e *Engine) addAggregateMediaTypeSupplements(
ctx context.Context,
embedding []float32,
excludeIDs []string,
filter catalog.AccessFilter,
candidates []ScoredItem,
genreMap map[string][]string,
limit int,
) ([]ScoredItem, map[string][]string, map[string]string) {
mediaTypes, err := e.repo.GetItemMediaTypes(ctx, scoredItemIDs(candidates))
if err != nil {
return candidates, genreMap, map[string]string{}
}
floor := mediaTypeFloor(limit)
changed := false
for _, mediaType := range aggregateSupplementMediaTypes {
if countMediaType(candidates, mediaTypes, mediaType) >= floor {
continue
}
extra, extraGenres, err := e.repo.FindTasteProfileCandidatesByMediaType(ctx, embedding, excludeIDs, nil, limit, filter, mediaType)
if err != nil || len(extra) == 0 {
continue
}
candidates = mergeScoredCandidates(candidates, extra)
for id, genres := range extraGenres {
genreMap[id] = genres
}
changed = true
}
if changed {
if refreshed, err := e.repo.GetItemMediaTypes(ctx, scoredItemIDs(candidates)); err == nil {
mediaTypes = refreshed
}
}
return candidates, genreMap, mediaTypes
}
func applyMediaTypeFloor(items []ScoredItem, candidates []ScoredItem, mediaTypes map[string]string) []ScoredItem {
if len(items) == 0 || len(candidates) == 0 || len(mediaTypes) == 0 {
return items
}
floor := mediaTypeFloor(len(items))
for _, mediaType := range aggregateSupplementMediaTypes {
if countMediaType(candidates, mediaTypes, mediaType) == 0 {
continue
}
items = ensureMediaTypeFloor(items, candidates, mediaTypes, mediaType, floor)
}
return items
}
func ensureMediaTypeFloor(items []ScoredItem, candidates []ScoredItem, mediaTypes map[string]string, mediaType string, floor int) []ScoredItem {
if floor <= 0 || countMediaType(items, mediaTypes, mediaType) >= floor {
return items
}
selected := make(map[string]struct{}, len(items))
for _, item := range items {
selected[item.MediaItemID] = struct{}{}
}
needed := floor - countMediaType(items, mediaTypes, mediaType)
replacements := make([]ScoredItem, 0, needed)
for _, candidate := range candidates {
if len(replacements) >= needed {
break
}
if mediaTypes[candidate.MediaItemID] != mediaType {
continue
}
if _, ok := selected[candidate.MediaItemID]; ok {
continue
}
replacements = append(replacements, candidate)
selected[candidate.MediaItemID] = struct{}{}
}
if len(replacements) == 0 {
return items
}
remove := make(map[string]struct{}, len(replacements))
for i := len(items) - 1; i >= 0 && len(remove) < len(replacements); i-- {
if mediaTypes[items[i].MediaItemID] == mediaType {
continue
}
remove[items[i].MediaItemID] = struct{}{}
}
if len(remove) < len(replacements) {
return items
}
mixed := make([]ScoredItem, 0, len(items))
for _, item := range items {
if _, ok := remove[item.MediaItemID]; ok {
continue
}
mixed = append(mixed, item)
}
mixed = append(mixed, replacements...)
sortScoredItems(mixed)
return mixed
}
func mediaTypeFloor(limit int) int {
if limit <= 0 {
return 0
}
floor := (limit + aggregateMediaTypeFloorDivisor - 1) / aggregateMediaTypeFloorDivisor
if floor < 1 {
return 1
}
return floor
}
func countMediaType(items []ScoredItem, mediaTypes map[string]string, mediaType string) int {
count := 0
for _, item := range items {
if mediaTypes[item.MediaItemID] == mediaType {
count++
}
}
return count
}
func scoredItemIDs(items []ScoredItem) []string {
ids := make([]string, 0, len(items))
for _, item := range items {
ids = append(ids, item.MediaItemID)
}
return ids
}
func mergeScoredCandidates(base []ScoredItem, extra []ScoredItem) []ScoredItem {
seen := make(map[string]struct{}, len(base)+len(extra))
merged := make([]ScoredItem, 0, len(base)+len(extra))
for _, item := range base {
if _, ok := seen[item.MediaItemID]; ok {
continue
}
seen[item.MediaItemID] = struct{}{}
merged = append(merged, item)
}
for _, item := range extra {
if _, ok := seen[item.MediaItemID]; ok {
continue
}
seen[item.MediaItemID] = struct{}{}
merged = append(merged, item)
}
sortScoredItems(merged)
return merged
}
func sortScoredItems(items []ScoredItem) {
sort.SliceStable(items, func(i, j int) bool {
if items[i].Score != items[j].Score {
return items[i].Score > items[j].Score
}
return items[i].MediaItemID < items[j].MediaItemID
})
}
// buildColdStartRows generates non-personalized rows.
func (e *Engine) buildColdStartRows(ctx context.Context) ([]ForYouRow, error) {
popular, _ := e.repo.GetPopularItems(ctx, 30, 20)
recentlyAdded, _ := e.repo.GetRecentlyAddedItems(ctx, 14, 20)
topRated, _ := e.repo.GetTopRatedItems(ctx, 5, 20)
genreSamplers := make(map[string][]ScoredItem)
topGenres, _ := e.repo.GetTopGenres(ctx, 5)
for _, genre := range topGenres {
items, _ := e.repo.GetGenreSamplerItems(ctx, genre, 20)
if len(items) > 0 {
genreSamplers[genre] = items
}
}
return buildColdStartRows(popular, recentlyAdded, topRated, genreSamplers), nil
}
// BecauseYouWatched returns items similar to a specific item the user has
// watched. Blends embedding similarity (70%) with co-watch data (30%).
func (e *Engine) BecauseYouWatched(ctx context.Context, userID int, profileID string, sourceItemID string, limit int) ([]ScoredItem, error) {
embedding, err := e.repo.GetEmbedding(ctx, sourceItemID)
if err != nil {
return nil, fmt.Errorf("get embedding for item %s: %w", sourceItemID, err)
}
if embedding == nil {
return nil, nil
}
// Constrain to the source item's media type so "Because you watched" never
// mixes movies, series, and audiobooks in one rail.
sourceMeta, _ := e.repo.GetItemMetadata(ctx, sourceItemID)
sourceType := ""
if sourceMeta != nil {
sourceType = sourceMeta.Type
}
// Get embedding-based candidates (3x for MMR).
embCandidates, err := e.repo.FindSimilar(ctx, embedding, []string{sourceItemID}, sourceType, limit*3)
if err != nil {
return nil, fmt.Errorf("find similar for because watched: %w", err)
}
// Get co-watch neighbors.
cowatchPairs, _ := e.repo.GetCowatchNeighbors(ctx, sourceItemID, limit*3)
cowatchMap := make(map[string]float64, len(cowatchPairs))
for _, p := range cowatchPairs {
cowatchMap[p.SimilarItemID] = p.JaccardScore
}
// Blend scores.
blended := blendScores(embCandidates, cowatchMap, 0.7, 0.3)
// Apply MMR re-ranking.
candidateIDs := make([]string, len(blended))
for i, item := range blended {
candidateIDs[i] = item.MediaItemID
}
embMap, _ := e.repo.GetBatchEmbeddings(ctx, candidateIDs)
result := applyMMR(blended, embMap, e.mmrLambda(LambdaBecauseWatched), limit)
for i := range result {
if result[i].Reason == "" {
result[i].Reason = "because_you_watched"
}
}
watchedSet, err := e.watchedItemIDSet(ctx, userID, profileID)
if err != nil {
return nil, fmt.Errorf("get watched item IDs: %w", err)
}
return excludeScoredItems(result, watchedSet), nil
}
func excludeWatchedRows(rows []ForYouRow, watchedSet map[string]struct{}) []ForYouRow {
if len(rows) == 0 || len(watchedSet) == 0 {
return rows
}
filteredRows := make([]ForYouRow, 0, len(rows))
for _, row := range rows {
row.Items = excludeScoredItems(row.Items, watchedSet)
if len(row.Items) == 0 {
continue
}
filteredRows = append(filteredRows, row)
}
return filteredRows
}