Files
silo-server/internal/literaryworks/service.go
57fbc1e0b4 perf(literaryworks): cut Postgres load on ebook↔audiobook auto-linking (#253)
* perf(literaryworks): cut Postgres load on ebook↔audiobook auto-linking

Rescans and candidate matching were driving high Postgres CPU on book
libraries.

- AutoLinkContent now checks literary_work_items with a cheap indexed
  lookup before GetMatchItem, so unchanged already-linked books skip the
  heavy triple-lateral query on every rescan.
- ListMatchCandidates is split into an indexable ID-selection phase
  (title / provider EXISTS / series EXISTS) and a hydration phase, so the
  per-row lateral aggregates run only for the LIMIT candidates kept rather
  than for every opposite-format book.
- Add a partial LOWER(title) index scoped to ebook/audiobook so the OR of
  title/provider/series filters can be driven entirely by indexes.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* fix(migrations): drop invalid leftover index before concurrent rebuild

An interrupted CREATE INDEX CONCURRENTLY (cancel/restart) can leave an
INVALID idx_media_items_books_title_lower; IF NOT EXISTS would then skip
the rebuild while Goose records success. Add the preflight invalid-index
drop used by the repo's other concurrent-index migrations.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
2026-07-01 23:21:25 -04:00

275 lines
7.7 KiB
Go

package literaryworks
import (
"context"
"crypto/sha256"
"encoding/hex"
"mime"
"path/filepath"
"sort"
"strings"
"github.com/Silo-Server/silo-server/internal/catalog"
)
type Service struct {
repo *Repository
}
func NewService(repo *Repository) *Service {
return &Service{repo: repo}
}
func (s *Service) GetWork(ctx context.Context, workID string, filter catalog.AccessFilter) (*DetailResponse, error) {
if s == nil || s.repo == nil {
return nil, ErrWorkNotFound
}
work, items, err := s.repo.GetWorkWithItems(ctx, workID, filter)
if err != nil {
return nil, err
}
resp := &DetailResponse{
WorkID: work.WorkID,
WorkTitle: work.CanonicalTitle,
Metadata: WorkMetadata{
Description: work.Description,
Genres: work.Genres,
Publisher: work.Publisher,
},
}
if work.PublishedDate != nil {
resp.Metadata.PublishedDate = work.PublishedDate.Format("2006-01-02")
}
for _, item := range items {
resp.Formats = append(resp.Formats, FormatResponse{
Type: item.FormatType,
ContentID: item.ContentID,
LibraryID: item.LibraryID,
AvailableFiles: filesToResponse(item.Files),
Progress: item.Progress,
})
}
return resp, nil
}
func (s *Service) ListCandidates(ctx context.Context, contentID string, limit int) ([]Candidate, error) {
if s == nil || s.repo == nil {
return nil, ErrWorkNotFound
}
source, err := s.repo.GetMatchItem(ctx, contentID)
if err != nil {
return nil, err
}
targets, err := s.repo.ListMatchCandidates(ctx, source.MatchItem, limit*5)
if err != nil {
return nil, err
}
candidates := make([]Candidate, 0, len(targets))
for _, target := range targets {
candidate := ScoreCandidate(source.MatchItem, target.MatchItem)
candidate.TargetWorkID = target.WorkID
if candidate.Score >= 0.75 {
candidates = append(candidates, candidate)
}
}
sort.SliceStable(candidates, func(i, j int) bool {
if candidates[i].Score == candidates[j].Score {
return candidates[i].TargetContentID < candidates[j].TargetContentID
}
return candidates[i].Score > candidates[j].Score
})
if limit > 0 && len(candidates) > limit {
candidates = candidates[:limit]
}
return candidates, nil
}
func (s *Service) LinkItems(ctx context.Context, workID string, contentIDs []string) (string, error) {
if s == nil || s.repo == nil {
return "", ErrWorkNotFound
}
contentIDs = compactContentIDs(contentIDs)
if len(contentIDs) == 0 {
return "", ErrWorkNotFound
}
items := make([]MatchItemWithWork, 0, len(contentIDs))
for _, contentID := range contentIDs {
item, err := s.repo.GetMatchItem(ctx, contentID)
if err != nil {
return "", err
}
items = append(items, item)
}
if strings.TrimSpace(workID) == "" {
existing, err := s.repo.GetFirstWorkIDForContentIDs(ctx, contentIDs)
if err != nil {
return "", err
}
workID = existing
}
if strings.TrimSpace(workID) == "" {
workID = generatedWorkID(items[0].MatchItem)
if _, err := s.repo.CreateWork(ctx, CreateWorkParams{
WorkID: workID,
CanonicalTitle: items[0].Title,
SortTitle: items[0].Title,
NormalizedTitle: normalizeKey(items[0].Title),
PrimaryAuthorKey: personKey(items[0].Authors),
Publisher: items[0].Publisher,
}); err != nil {
return "", err
}
}
linkItems := make([]LinkItemParams, 0, len(items))
for _, item := range items {
linkItems = append(linkItems, LinkItemParams{
ContentID: item.ContentID,
FormatType: item.Type,
LinkSource: LinkManual,
Confidence: 1,
})
}
if err := s.repo.LinkItems(ctx, workID, linkItems); err != nil {
return "", err
}
return workID, nil
}
func (s *Service) AutoLinkContent(ctx context.Context, contentID string) (string, bool, error) {
if s == nil || s.repo == nil {
return "", false, ErrWorkNotFound
}
// Cheap guard first: an item already linked to a work needs no candidate
// scan. Rescans re-visit every unchanged book, so skipping the expensive
// GetMatchItem + ListMatchCandidates path here is what keeps Postgres idle
// on steady-state rescans instead of rebuilding lateral aggregates per book.
if workID, err := s.repo.GetFirstWorkIDForContentIDs(ctx, []string{contentID}); err != nil {
return "", false, err
} else if strings.TrimSpace(workID) != "" {
return workID, false, nil
}
source, err := s.repo.GetMatchItem(ctx, contentID)
if err != nil {
return "", false, err
}
if strings.TrimSpace(source.WorkID) != "" {
return source.WorkID, false, nil
}
targets, err := s.repo.ListMatchCandidates(ctx, source.MatchItem, 100)
if err != nil {
return "", false, err
}
var best Candidate
var bestTarget MatchItemWithWork
for _, target := range targets {
candidate := ScoreCandidate(source.MatchItem, target.MatchItem)
if candidate.Score > best.Score {
best = candidate
bestTarget = target
}
}
if best.Score < AutoLinkThreshold || best.TargetContentID == "" {
return "", false, nil
}
workID := strings.TrimSpace(bestTarget.WorkID)
if workID == "" {
workID = generatedWorkID(source.MatchItem)
if _, err := s.repo.CreateWork(ctx, CreateWorkParams{
WorkID: workID,
CanonicalTitle: source.Title,
SortTitle: source.Title,
NormalizedTitle: normalizeKey(source.Title),
PrimaryAuthorKey: personKey(source.Authors),
Publisher: source.Publisher,
}); err != nil {
return "", false, err
}
}
items := []LinkItemParams{{
ContentID: source.ContentID,
FormatType: source.Type,
LinkSource: best.LinkSource,
Confidence: best.Score,
}}
if bestTarget.WorkID == "" {
items = append(items, LinkItemParams{
ContentID: bestTarget.ContentID,
FormatType: bestTarget.Type,
LinkSource: best.LinkSource,
Confidence: best.Score,
})
}
if err := s.repo.LinkItems(ctx, workID, items); err != nil {
return "", false, err
}
return workID, true, nil
}
func (s *Service) UnlinkItem(ctx context.Context, workID, contentID string) error {
if s == nil || s.repo == nil {
return ErrWorkNotFound
}
return s.repo.UnlinkItem(ctx, workID, contentID)
}
func (s *Service) ConfirmMatch(ctx context.Context, sourceContentID, targetContentID string, userID int) (string, error) {
if s == nil || s.repo == nil {
return "", ErrWorkNotFound
}
workID, err := s.LinkItems(ctx, "", []string{sourceContentID, targetContentID})
if err != nil {
return "", err
}
if err := s.repo.RecordDecision(ctx, sourceContentID, targetContentID, DecisionConfirmed, userID); err != nil {
return "", err
}
return workID, nil
}
func (s *Service) IgnoreMatch(ctx context.Context, sourceContentID, targetContentID string, userID int) error {
if s == nil || s.repo == nil {
return ErrWorkNotFound
}
return s.repo.RecordDecision(ctx, sourceContentID, targetContentID, DecisionIgnored, userID)
}
func filesToResponse(files []WorkFile) []FileResponse {
out := make([]FileResponse, 0, len(files))
for _, f := range files {
ext := strings.TrimPrefix(strings.ToLower(filepath.Ext(f.FilePath)), ".")
out = append(out, FileResponse{
FileID: f.FileID,
OriginalName: filepath.Base(f.FilePath),
Format: ext,
MIMEType: mime.TypeByExtension("." + ext),
Size: f.Size,
DurationSeconds: f.DurationSeconds,
})
}
return out
}
func compactContentIDs(contentIDs []string) []string {
seen := map[string]struct{}{}
out := make([]string, 0, len(contentIDs))
for _, contentID := range contentIDs {
contentID = strings.TrimSpace(contentID)
if contentID == "" {
continue
}
if _, ok := seen[contentID]; ok {
continue
}
seen[contentID] = struct{}{}
out = append(out, contentID)
}
return out
}
func generatedWorkID(item MatchItem) string {
title := normalizeKey(item.Title)
author := personKey(item.Authors)
sum := sha256.Sum256([]byte(title + "|" + author))
return "work-" + hex.EncodeToString(sum[:])[:16]
}