* docs: define ebook architecture matching audiobooks * docs: plan ebook audiobook-parity implementation * feat: add ebook scanner parser foundation * fix: harden ebook scanner foundation * fix: handle ebook isbn labels * fix: guard ebook subtree scans * feat: scan ebook libraries in core * fix: preserve ebook scan people credits * fix: refresh ebook scan metadata safely * feat: persist ebook series membership * test: cover ebook series persistence decisions * fix: address ebook scanner PR review * docs: clarify ebook foundation PR scope * feat: add ebook metadata enricher * fix: harden ebook poster cache * feat: wire ebook metadata sync task * feat: expose ebook library metadata setup * feat: add ebook catalog scope support * feat: add ebook detail view * feat: label ebook file versions by format * feat: use file-size copy for downloads * feat: use file language in download dialog * test: cover ebook detail authors and downloads * fix: drop narrator credits from ebook scanner merges * fix: align ebook collection filters with book media * fix: drop asin provider ids from ebook enrichment * fix: force ebook people refresh for stale narrators * chore: omit ebook planning docs from branch * feat: add ebook detail related content * feat: add ebook reader file entrypoint * feat: render ebooks with foliate reader * feat: persist ebook reader progress * feat: add ebook reader controls * feat: extract ebook pdf metadata * feat: favor scanner isbn during ebook enrichment * feat: extract fbz ebook metadata * feat: count cbz ebook pages * feat: show ebook file page counts * feat: show ebook download summaries * feat: switch ebook reader files * feat: prefer epub for ebook read action * feat: surface ebook reader progress * feat: sync ebook reader progress cache * feat: hide ebook read action for unsupported files * feat: filter ebook reader file selector * fix: serve fbz ebook archives with reader mime type * fix: detect fbz ebooks from compound filename * fix: authorize fbz ebooks from compound filename * fix: scope ebook catalog facets * fix: reject narrator queries for ebooks * fix: build ebook recommendation text from authors * fix: include ebooks in embedding eligibility * fix: include ebooks in recommendation media mix * fix: include ebooks in recently added recommendations * feat: include ebook progress in recommendation signals * feat: include ebooks in continue watching sections * feat: include ebooks in catalog progress metrics * fix: read ebook isbn from epub metadata * fix: filter ebook asin provider aliases * fix: fall back from unsupported ebook reader files * fix: sort ebook catalogs by reader progress * fix: filter ebook catalogs by reader progress * fix: include ebooks in last watched catalog filters * feat: reflect ebook reader progress in item user state * feat: share ebook progress state across item surfaces * feat: report ebook scan progress * fix: include ebook activity in recommendations * fix: expose ebook reader progress on item detail * fix: support ebook subtree scans * fix: honor profile header for ebook item progress * fix: add ebook library default sections * fix: route ebook continue cards to reader * fix: hide watched toggle for ebooks * fix: route ebook watch tonight cards to reader * fix: route ebook hero actions to reader * fix: detect archive ebook reader formats by filename * feat: cache embedded ebook covers during scan * fix: encode ebook hero reader links * fix: persist non-epub ebook reader progress * fix: scope narrator catalog badges to audiobooks * fix: merge ebook reader progress during item repair * fix: label ebook progress filters as read * fix: show ebook related rails as book covers * fix: remove txt ebook reader support * fix: reject txt ebook reader files * fix: label ebook advanced filters as read * fix: label ebook personalized sorts as read * fix: remove plain text reader loader path * test: cover ebook unread catalog rules * fix: preserve ebook reader library context * fix: link ebook genres with library scope * fix: encode related rail item links * fix: encode catalog card item links * fix: encode hero and continue item links * fix: encode watch tonight item links * fix: encode recommendation and search item links * test: cover ebook scan format set * fix: label ebook search results clearly * fix: make global search prompt media neutral * fix: encode catalog read API ids * fix: encode item API ids * fix: include ebook reader vendor in docker build * fix: make ebook reader build clean * fix: clean ebook embedded descriptions * docs: plan ebook reader shell parity * feat: add ebook reader shell controls * fix: widen ebook scrolled reader flow * fix: remove scrolled reader content width cap * docs: plan ebook reader full parity * feat: persist ebook reader config * feat: add ebook annotations and bookmarks * feat: add ebook reader tools and aids * feat: add ebook advanced reader settings * fix: keep ebook reader panel in viewport * fix: use foliate sizing units for ebook scroll flow * fix: keep ebook settings controls readable * fix: simplify ebook reader settings controls * feat(ebooks): extract local covers during scan (#98) * feat(ebooks): extract local covers during scan * fix(ebooks): read nullable poster paths during cover scan * fix(catalog): coalesce nullable media artwork fields * fix(ebooks): group sibling formats by book identity * fix(ebooks): tolerate legacy ebook metadata encodings * fix(ebooks): decode PDF hex metadata strings * fix(ebooks): harden local cover extraction and format grouping Address review findings on the local cover scan: - Restrict generic sidecar covers (cover.jpg, folder.png, ...) to single-book directories, always accept images named after the book file, and apply exactly one cover per reconcile with sidecar taking precedence over the embedded cover. - Replace the read-then-write poster update with an atomic conditional UPDATE (ItemRepository.SetLocalPoster) so provider/admin artwork is never clobbered by concurrent writers, and refresh locally owned posters when the extracted cover bytes change (thumbhash compare). - Preserve UTF-8 PDF Info strings (including a UTF-8 BOM) instead of forcing everything through Windows-1252; the cp1252 fallback now only applies to non-UTF-8 bytes. - Select EPUB covers by manifest media-type with properties="cover-image" outranking the EPUB2 meta name="cover" id, so XHTML cover pages no longer shadow the real image. - Order CBZ pages naturally (2.jpg before 10.jpg, ch2/ before ch10/) when picking the cover page, via a single O(n) min-scan. - Bump the ebook content group key scheme to version 2 and reprocess rows written under older versions so pre-existing libraries gain sibling-format grouping instead of accumulating duplicates. - Group different formats only (a same-format sibling with colliding sparse metadata stays a separate item) and stop a joining sibling's embedded metadata from overwriting a provider-matched item. - Decode any IANA-labelled OPF/FB2 XML charset (windows-1251, koi8-r, shift_jis, ...) via x/net/html/charset, and wire the charset reader into FB2 parsing which previously had none. - Strip the full .fb2.zip double extension from filename-derived titles and group keys. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: rxwatcher <rxwatcher@users.noreply.github.com> Co-authored-by: Quick <31828688+Quick104@users.noreply.github.com> Co-authored-by: Claude Fable 5 <noreply@anthropic.com> * feat(ebooks): add reader profiles and ruler (#99) * feat(ebooks): extract local covers during scan * fix(ebooks): read nullable poster paths during cover scan * fix(catalog): coalesce nullable media artwork fields * fix(ebooks): group sibling formats by book identity * fix(ebooks): tolerate legacy ebook metadata encodings * fix(ebooks): decode PDF hex metadata strings * feat(ebooks): add reader profiles and ruler * fix(ebooks): address reader ruler and profile review findings - skip renderer setStyles/render when computed styles and attributes are unchanged, so ruler position updates no longer re-style the book view - drag the ruler via a local draft that commits on release, with the surface rect cached at pointer-down - migrate font values persisted before the generic stacks (Inter, Georgia, Merriweather, legacy serif) so the font select never renders blank, with a Custom fallback option for unknown values - make the ruler band click-through and move dragging to a dedicated keyboard-accessible slider handle so links and text selection keep working under the band - share font stacks between options and profiles via READER_FONT_STACKS - surface the active reading profile, move presets to the top of the settings panel, and drop the redundant profile button aria-labels Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(ebooks): resolve prefer-const lint error in readest document lib `pnpm run lint` failed on the branch because `direction` is never reassigned in getDirection; split the destructure so only the reassigned `writingMode` stays mutable. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: rxwatcher <rxwatcher@users.noreply.github.com> Co-authored-by: Quick <31828688+Quick104@users.noreply.github.com> Co-authored-by: Claude Fable 5 <noreply@anthropic.com> * Merge branch 'main' into work/ebooks-reader-base Brings the ebook integration branch up to date with main (audiobook library redesign, continue-watching rework and card affordances, quic-go bump, jellycompat fixes). Conflict resolutions favor main's generalized mechanisms and register ebooks with them: - media scope validation goes through IsValidMediaScope (now including "ebook" alongside main's "video" group scope), in Go and in the web filter/search types - continue-watching uses main's typed rails; reading-type sections pull resume points from ebook_reader_progress and the ebook library default section is wired to ContinueTypeConfig(ContinueTypeReading) - item_repo keeps main's derived select-list machinery (itemColumnExpr) and both poster accessors (GetPoster/SetLocalPoster for ebook covers, GetPosterPath for audiobook covers) - web cards/hero/watch-tonight adopt main's buildMediaPlayHref helpers, which now route ebooks to /reader/ebook and encode content ids; ebook affordances (BookOpen icon, Read verb, percent-read subtitle) carry over onto main's reworked components - LibraryForm ebook support ported into main's refactored useLibraryForm/libraryTypes modules Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(docker): copy foliate-js vendor into Dockerfile.dev frontend stage foliate-js is a file:vendor/foliate-js dependency, so pnpm install needs the vendor directory before the lockfile install layer. The production Dockerfile already copies it; the dev image was missed, breaking make dev-deploy with ENOENT on /app/web/vendor/foliate-js. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * feat(ebooks): render Continue Reading sections as upright poster cards All-ebook continue sections previously fell through to the horizontal 16:9 wide card; include ebooks in the poster-variant check so book covers render in their natural 2:3 framing. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(ui): stop related-rail highlight ring clipping on detail pages Move the current-item ring onto the cover artwork with a themed ring-offset color (matching the sidebar profile highlight) and give the scroll container top headroom so the ring is not cut off by overflow-x-auto. Applies to both ebook and audiobook detail rails. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(scanner): harden ebook scanning against data loss and bad metadata - Reconcile missing ebook files like video/audio, with real per-root walk failure tracking (failed/unmounted roots are excluded from deletion), symlinked-root support via the shared logical walker, and the empty-root cleanup allowance before any destructive reconciliation. - Create ebook items as 'pending' so enrichment can promote them to 'matched' (backfill migration included), and protect matched items from re-scan clobbering: title/year skipped, people/series fill-empty only. - PDF metadata: scan head + tail windows (non-linearized PDFs keep the Info dict at the end), require proper key delimiters, head values win. - Cap plain .fb2 reads like .fbz entries; drop .md as an ebook format. - gofmt internal/scanner/audiobook.go (pre-existing drift). Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(ebooks): make enrichment failures non-terminal with dedicated backoff state - Provider errors now record a failure (capped retries) instead of stamping last_refreshed, which permanently excluded items after transient outages. - Unconfigured metadata chains and the scan-window membership race skip the item without stamping or burning a retry. - Failure tracking moves to a new ebook_enrichment_state table, decoupling it from media_items.refresh_failures (shared with metadata refresh debt). - Preserve non-author people credits when persisting enrichment results. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(catalog): gate ebook progress on hidden history and centralize threshold - Apply user_history_hidden_items gating (video semantics) to the ebook watched/in-progress filters, progress sort plan, and Continue Reading. - Continue Reading pages past dismissed items via the shared collector and dedupes items across pages (also fixes the video path's latent exposure). - Centralize the 0.9 finished threshold as models.EbookFinishedProgressThreshold with a single SQL-interpolated mirror in catalog. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(recommendations): correct watcher counting and wire ebook taste signals - itemWatchersQuery dedupes to distinct (watcher, item) rows so one binge-watcher can no longer satisfy minWatchers; the eligibility floor now counts distinct accounts rather than profiles. - Hidden-history gating on GetEbookReaderProgressForUser (signal reader). - Ebook reading produces canonical implicit taste signals (weighted like the equivalent movie progress ratio); ebooks join taste-seed candidates. - Stale GetRecentlyAddedItems doc comment corrected. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(api): harden ebook reader endpoints and serve a Content-Security-Policy - Serve a CSP on all SPA HTML responses: blob/srcdoc book iframes inherit it, so script-src 'self' 'wasm-unsafe-eval' blocks script execution from malicious book content (sandbox alone is defeated by the WebKit allow-scripts requirement). Threat model documented on the constant. - X-Content-Type-Options: nosniff on frontend, jellycompat, and ebook file responses; MIME resolution can no longer fall through to octet-stream for an admitted ebook file. - Annotation PATCH: presence-aware field semantics (absent keeps, present sets/clears), invariant re-validation on the merged row, and an atomic SELECT ... FOR UPDATE read-merge-write. - Request size caps (413) on progress/config/annotation writes; Content-Disposition via mime.FormatMediaType; hidden-history gating in the shared ebook progress lister; FK-cascade indexes for reader tables. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * feat(api): native read-state endpoints for ebooks - POST/DELETE /watched/{id} accepts ebook content IDs: mark read upserts progress 1.0 preserving the reader's file/location (or picks the preferred reader file for never-opened books); mark unread mirrors video unwatch semantics and deletes the progress row. - /history/remove accepts ebooks: hides via user_history_hidden_items without touching the reading position (hidden != unread; next reading activity resurfaces the book, mirroring video re-watch). - Access-filter checks match the video branch; shared logic lives in ebook_read_state.go. Sort metrics/user-state thresholds use the shared constant; profile-header fallback deduplicated. Clients: response is {type: "ebook", affected_count: 1, played: bool}; the existing watched SSE event fires. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * fix(web): harden the ebook reader UI - Open-flow race: cancellation checked after every await with full stale-run teardown (no wrong-file progress saves, no leaked views/blob URLs); book.destroy() on cleanup. - Progress: monotonic stale-response guard; visibilitychange flush uses the refresh-capable client, pagehide uses keepalive; per-book cross-format progress documented as deliberate. - Settings: side effects out of the setState updater; local edits no longer clobbered by late server config; pending saves flushed on unmount/pagehide. - TTS: generation token so Stop actually stops (Chromium/Firefox synthetic events); Media Session uninstalled on unmount. - External book links: http(s) only, opened with noopener,noreferrer. - apiBlob 512 MiB guard with a user-facing error; fraction bookmarks navigable; search-result key collisions fixed; dead e-ink code removed; getLibrarySortRelevanceScope deduplicated; md format dropped. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * feat(web): mark read/unread affordances for ebooks - Item detail gets a Mark Read/Unread button; card menus drop the ebook gate and share type-aware labels/toasts (also dedupes audiobook wording). - Watched-state invalidation includes the reader progress query key so the Continue button and percent refresh after toggling. - Continue Reading dismiss copy for ebooks; dismissal path now URL-encodes item IDs (ebook content IDs can contain reserved characters). Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> * docs: record the PR #124 review and hardening pass Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> --------- Co-authored-by: rxwatcher <rxwatcher@users.noreply.github.com> Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
437 lines
13 KiB
Go
437 lines
13 KiB
Go
// Package imagecache downloads images from URLs, generates sized variants,
|
|
// computes thumbhashes, and uploads all variants to S3.
|
|
package imagecache
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"net"
|
|
"net/http"
|
|
"net/netip"
|
|
"net/url"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
|
|
"github.com/Silo-Server/silo-server/internal/imageutil"
|
|
"github.com/Silo-Server/silo-server/internal/metadata"
|
|
)
|
|
|
|
const (
|
|
maxDownloadBytes = 10 * 1024 * 1024 // 10 MB
|
|
downloadTimeout = 30 * time.Second
|
|
)
|
|
|
|
// ObjectPutter is the S3 interface required by Cacher.
|
|
type ObjectPutter interface {
|
|
PutObject(ctx context.Context, bucket, key string, data []byte) error
|
|
Bucket() string
|
|
}
|
|
|
|
// ImageURLResolver resolves plugin:// paths to HTTP URLs.
|
|
type ImageURLResolver interface {
|
|
ResolveImageURL(ctx context.Context, path string, variant string) string
|
|
}
|
|
|
|
// CacheRequest describes a single image to cache. For season posters and
|
|
// episode stills, ContentID is the parent series's provider ID and the
|
|
// SeasonNumber / EpisodeNumber fields scope the S3 key so siblings do not
|
|
// collide. Both pointers are nil for item-level images.
|
|
type CacheRequest struct {
|
|
SourceURL string
|
|
ProviderID string
|
|
ContentType string // "movies" or "series"
|
|
ContentID string
|
|
ImageType metadata.ImageType
|
|
SeasonNumber *int
|
|
EpisodeNumber *int
|
|
ImageResolver ImageURLResolver // optional; used when SourceURL is a plugin:// path
|
|
}
|
|
|
|
// CacheResult is returned by Cache on success.
|
|
type CacheResult struct {
|
|
BasePath string // S3 key prefix, e.g. "tmdb/movies/550/poster"
|
|
Thumbhash string // base64-encoded
|
|
Ext string // file extension including dot (e.g. ".jpg", ".png")
|
|
}
|
|
|
|
// Cacher downloads and stores image variants to S3.
|
|
type Cacher struct {
|
|
s3 ObjectPutter
|
|
httpClient *http.Client
|
|
enforcePublicURLs bool
|
|
}
|
|
|
|
// New creates a new Cacher backed by the given ObjectPutter.
|
|
func New(s3 ObjectPutter) *Cacher {
|
|
return &Cacher{s3: s3, httpClient: newSecureHTTPClient(), enforcePublicURLs: true}
|
|
}
|
|
|
|
func newWithHTTPClient(s3 ObjectPutter, client *http.Client) *Cacher {
|
|
if client == nil {
|
|
client = http.DefaultClient
|
|
}
|
|
return &Cacher{s3: s3, httpClient: client}
|
|
}
|
|
|
|
// CacheImage implements metadata.ImageCacher using the internal Cache method.
|
|
func (c *Cacher) CacheImage(ctx context.Context, req metadata.CacheImageRequest) (*metadata.CacheImageResult, error) {
|
|
result, err := c.Cache(ctx, CacheRequest{
|
|
SourceURL: req.SourceURL,
|
|
ProviderID: req.ProviderID,
|
|
ContentType: req.ContentType,
|
|
ContentID: req.ContentID,
|
|
ImageType: req.ImageType,
|
|
SeasonNumber: req.SeasonNumber,
|
|
EpisodeNumber: req.EpisodeNumber,
|
|
})
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
return &metadata.CacheImageResult{
|
|
BasePath: result.BasePath,
|
|
Thumbhash: result.Thumbhash,
|
|
Ext: result.Ext,
|
|
}, nil
|
|
}
|
|
|
|
// CacheAudiobookCover is a thin convenience over CacheBytes specifically
|
|
// for the audiobook scanner. Avoids exporting the imagecache request
|
|
// struct to the scanner package (which would create an import cycle
|
|
// scanner -> imagecache -> metadata -> scanner). Stores under
|
|
// "local/audiobooks/{contentID}/poster/...".
|
|
func (c *Cacher) CacheAudiobookCover(ctx context.Context, data []byte, contentID string) (basePath string, ext string, thumbhash string, err error) {
|
|
res, err := c.CacheBytes(ctx, data, CacheRequest{
|
|
ProviderID: "local",
|
|
ContentType: "audiobooks",
|
|
ContentID: contentID,
|
|
ImageType: metadata.ImagePoster,
|
|
})
|
|
if err != nil {
|
|
return "", "", "", err
|
|
}
|
|
return res.BasePath, res.Ext, res.Thumbhash, nil
|
|
}
|
|
|
|
// CacheEbookCover stores an embedded ebook cover under
|
|
// "local/ebooks/{contentID}/poster/..." using the same poster variants as
|
|
// provider-hosted book artwork.
|
|
func (c *Cacher) CacheEbookCover(ctx context.Context, data []byte, contentID string) (basePath string, ext string, thumbhash string, err error) {
|
|
res, err := c.CacheBytes(ctx, data, CacheRequest{
|
|
ProviderID: "local",
|
|
ContentType: "ebooks",
|
|
ContentID: contentID,
|
|
ImageType: metadata.ImagePoster,
|
|
})
|
|
if err != nil {
|
|
return "", "", "", err
|
|
}
|
|
return res.BasePath, res.Ext, res.Thumbhash, nil
|
|
}
|
|
|
|
// CacheBytes performs the same variant generation, thumbhash, and S3 upload as
|
|
// Cache but starts from raw image bytes already in hand. Used by the
|
|
// audiobook scanner to push embedded M4B cover art into S3 without round-
|
|
// tripping through HTTP.
|
|
func (c *Cacher) CacheBytes(ctx context.Context, data []byte, req CacheRequest) (*CacheResult, error) {
|
|
if strings.TrimSpace(req.ProviderID) == "" {
|
|
return nil, fmt.Errorf("imagecache: provider ID is required")
|
|
}
|
|
if strings.TrimSpace(req.ContentType) == "" {
|
|
return nil, fmt.Errorf("imagecache: content type is required")
|
|
}
|
|
if strings.TrimSpace(req.ContentID) == "" {
|
|
return nil, fmt.Errorf("imagecache: content ID is required")
|
|
}
|
|
if len(data) == 0 {
|
|
return nil, fmt.Errorf("imagecache: image data is empty")
|
|
}
|
|
thumbhash, err := imageutil.Thumbhash(data)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("imagecache: thumbhash: %w", err)
|
|
}
|
|
widths := variantWidths(req.ImageType)
|
|
result, err := imageutil.GenerateVariants(data, widths)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("imagecache: generate variants: %w", err)
|
|
}
|
|
basePath := buildBasePath(req.ProviderID, req.ContentType, req.ContentID, req.ImageType, req.SeasonNumber, req.EpisodeNumber)
|
|
bucket := c.s3.Bucket()
|
|
var wg sync.WaitGroup
|
|
uploadErrs := make([]error, len(result.Variants))
|
|
for i, v := range result.Variants {
|
|
wg.Add(1)
|
|
go func(idx int, variant imageutil.Variant) {
|
|
defer wg.Done()
|
|
key := basePath + "/" + variant.Key + result.Ext
|
|
if err := c.s3.PutObject(ctx, bucket, key, variant.Data); err != nil {
|
|
uploadErrs[idx] = fmt.Errorf("imagecache: upload %s: %w", key, err)
|
|
}
|
|
}(i, v)
|
|
}
|
|
wg.Wait()
|
|
for _, err := range uploadErrs {
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
return &CacheResult{BasePath: basePath, Thumbhash: thumbhash, Ext: result.Ext}, nil
|
|
}
|
|
|
|
// Cache downloads the image at req.SourceURL, generates variants, computes a
|
|
// thumbhash, uploads all variants to S3, and returns the base path and thumbhash.
|
|
func (c *Cacher) Cache(ctx context.Context, req CacheRequest) (*CacheResult, error) {
|
|
if strings.TrimSpace(req.ProviderID) == "" {
|
|
return nil, fmt.Errorf("imagecache: provider ID is required")
|
|
}
|
|
if strings.TrimSpace(req.ContentType) == "" {
|
|
return nil, fmt.Errorf("imagecache: content type is required")
|
|
}
|
|
if strings.TrimSpace(req.ContentID) == "" {
|
|
return nil, fmt.Errorf("imagecache: content ID is required")
|
|
}
|
|
if req.EpisodeNumber != nil && req.SeasonNumber == nil {
|
|
return nil, fmt.Errorf("imagecache: episode number requires a season number")
|
|
}
|
|
|
|
url := req.SourceURL
|
|
|
|
// Resolve non-HTTP paths (e.g. plugin_id://path) via the resolver.
|
|
if !strings.HasPrefix(url, "http://") && !strings.HasPrefix(url, "https://") {
|
|
if req.ImageResolver == nil {
|
|
return nil, fmt.Errorf("imagecache: non-HTTP URL %q requires ImageResolver", url)
|
|
}
|
|
url = req.ImageResolver.ResolveImageURL(ctx, url, "original")
|
|
if url == "" {
|
|
return nil, fmt.Errorf("imagecache: resolver returned empty URL for %q", req.SourceURL)
|
|
}
|
|
}
|
|
|
|
data, err := c.downloadImage(ctx, url)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("imagecache: download %s: %w", url, err)
|
|
}
|
|
|
|
// Compute thumbhash from the original downloaded data (JPEG/PNG) before
|
|
// converting to WebP, since Go's image.Decode doesn't support WebP.
|
|
thumbhash, err := imageutil.Thumbhash(data)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("imagecache: thumbhash: %w", err)
|
|
}
|
|
|
|
widths := variantWidths(req.ImageType)
|
|
|
|
result, err := imageutil.GenerateVariants(data, widths)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("imagecache: generate variants: %w", err)
|
|
}
|
|
|
|
basePath := buildBasePath(req.ProviderID, req.ContentType, req.ContentID, req.ImageType, req.SeasonNumber, req.EpisodeNumber)
|
|
bucket := c.s3.Bucket()
|
|
|
|
// Upload all variants concurrently.
|
|
var wg sync.WaitGroup
|
|
uploadErrs := make([]error, len(result.Variants))
|
|
for i, v := range result.Variants {
|
|
wg.Add(1)
|
|
go func(idx int, variant imageutil.Variant) {
|
|
defer wg.Done()
|
|
key := basePath + "/" + variant.Key + result.Ext
|
|
if err := c.s3.PutObject(ctx, bucket, key, variant.Data); err != nil {
|
|
uploadErrs[idx] = fmt.Errorf("imagecache: upload %s: %w", key, err)
|
|
}
|
|
}(i, v)
|
|
}
|
|
wg.Wait()
|
|
|
|
for _, err := range uploadErrs {
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
|
|
return &CacheResult{
|
|
BasePath: basePath,
|
|
Thumbhash: thumbhash,
|
|
Ext: result.Ext,
|
|
}, nil
|
|
}
|
|
|
|
// variantWidths returns the resize widths for the given image type.
|
|
func variantWidths(t metadata.ImageType) []int {
|
|
switch t {
|
|
case metadata.ImagePoster:
|
|
return []int{500, 300}
|
|
case metadata.ImageBackdrop:
|
|
return []int{1920, 1280, 300}
|
|
case metadata.ImageLogo:
|
|
return []int{500}
|
|
case metadata.ImageStill:
|
|
return []int{500, 300}
|
|
default:
|
|
return []int{500, 300}
|
|
}
|
|
}
|
|
|
|
// buildBasePath constructs the S3 key prefix for a given image. Season
|
|
// posters and episode stills nest under their parent series so a single
|
|
// DeletePrefix on the series prefix cascades to all child images.
|
|
//
|
|
// item-level: {provider}/{type}/{id}/{imageType}
|
|
// season: {provider}/{type}/{id}/seasons/{n}/{imageType}
|
|
// episode: {provider}/{type}/{id}/seasons/{n}/episodes/{m}/{imageType}
|
|
func buildBasePath(providerID, contentType, contentID string, t metadata.ImageType, seasonNumber, episodeNumber *int) string {
|
|
imageTypeName := imageTypeName(t)
|
|
base := fmt.Sprintf("%s/%s/%s", providerID, contentType, contentID)
|
|
if seasonNumber != nil {
|
|
base = fmt.Sprintf("%s/seasons/%d", base, *seasonNumber)
|
|
if episodeNumber != nil {
|
|
base = fmt.Sprintf("%s/episodes/%d", base, *episodeNumber)
|
|
}
|
|
}
|
|
return base + "/" + imageTypeName
|
|
}
|
|
|
|
// imageTypeName returns the lowercase string name for an ImageType.
|
|
func imageTypeName(t metadata.ImageType) string {
|
|
switch t {
|
|
case metadata.ImagePoster:
|
|
return "poster"
|
|
case metadata.ImageBackdrop:
|
|
return "backdrop"
|
|
case metadata.ImageLogo:
|
|
return "logo"
|
|
case metadata.ImageStill:
|
|
return "still"
|
|
default:
|
|
return "unknown"
|
|
}
|
|
}
|
|
|
|
// downloadImage fetches the image at the given URL, enforcing size, timeout,
|
|
// and public-network limits.
|
|
func (c *Cacher) downloadImage(ctx context.Context, rawURL string) ([]byte, error) {
|
|
parsed, err := url.Parse(rawURL)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("parse URL: %w", err)
|
|
}
|
|
if c.enforcePublicURLs {
|
|
if err := validatePublicImageURL(parsed); err != nil {
|
|
return nil, err
|
|
}
|
|
}
|
|
ctx, cancel := context.WithTimeout(ctx, downloadTimeout)
|
|
defer cancel()
|
|
|
|
req, err := http.NewRequestWithContext(ctx, http.MethodGet, rawURL, nil)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("build request: %w", err)
|
|
}
|
|
|
|
client := c.httpClient
|
|
if client == nil {
|
|
client = newSecureHTTPClient()
|
|
}
|
|
resp, err := client.Do(req)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("http get: %w", err)
|
|
}
|
|
defer resp.Body.Close()
|
|
|
|
if resp.StatusCode != http.StatusOK {
|
|
return nil, fmt.Errorf("unexpected status %d", resp.StatusCode)
|
|
}
|
|
|
|
limited := io.LimitReader(resp.Body, maxDownloadBytes+1)
|
|
data, err := io.ReadAll(limited)
|
|
if err != nil {
|
|
return nil, fmt.Errorf("read body: %w", err)
|
|
}
|
|
if int64(len(data)) > maxDownloadBytes {
|
|
return nil, fmt.Errorf("image exceeds %d byte limit", maxDownloadBytes)
|
|
}
|
|
|
|
return data, nil
|
|
}
|
|
|
|
func newSecureHTTPClient() *http.Client {
|
|
transport := &http.Transport{
|
|
Proxy: nil,
|
|
DialContext: secureImageDialContext,
|
|
TLSHandshakeTimeout: 10 * time.Second,
|
|
}
|
|
return &http.Client{
|
|
Transport: transport,
|
|
CheckRedirect: func(req *http.Request, via []*http.Request) error {
|
|
if len(via) >= 10 {
|
|
return http.ErrUseLastResponse
|
|
}
|
|
return validatePublicImageURL(req.URL)
|
|
},
|
|
}
|
|
}
|
|
|
|
func validatePublicImageURL(u *url.URL) error {
|
|
if u == nil {
|
|
return fmt.Errorf("empty URL")
|
|
}
|
|
if u.Scheme != "http" && u.Scheme != "https" {
|
|
return fmt.Errorf("unsupported URL scheme %q", u.Scheme)
|
|
}
|
|
host := u.Hostname()
|
|
if host == "" {
|
|
return fmt.Errorf("URL host is required")
|
|
}
|
|
if addr, err := netip.ParseAddr(host); err == nil && !isPublicAddr(addr) {
|
|
return fmt.Errorf("private image host %q is not allowed", host)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func secureImageDialContext(ctx context.Context, network, address string) (net.Conn, error) {
|
|
host, port, err := net.SplitHostPort(address)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
addr, err := resolvePublicAddr(ctx, host)
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
dialer := &net.Dialer{Timeout: downloadTimeout}
|
|
return dialer.DialContext(ctx, network, net.JoinHostPort(addr.String(), port))
|
|
}
|
|
|
|
func resolvePublicAddr(ctx context.Context, host string) (netip.Addr, error) {
|
|
if addr, err := netip.ParseAddr(host); err == nil {
|
|
if isPublicAddr(addr) {
|
|
return addr, nil
|
|
}
|
|
return netip.Addr{}, fmt.Errorf("private image host %q is not allowed", host)
|
|
}
|
|
ips, err := net.DefaultResolver.LookupIPAddr(ctx, host)
|
|
if err != nil {
|
|
return netip.Addr{}, fmt.Errorf("resolve image host %q: %w", host, err)
|
|
}
|
|
for _, ip := range ips {
|
|
addr, ok := netip.AddrFromSlice(ip.IP)
|
|
if ok && isPublicAddr(addr) {
|
|
return addr, nil
|
|
}
|
|
}
|
|
return netip.Addr{}, fmt.Errorf("image host %q did not resolve to a public address", host)
|
|
}
|
|
|
|
func isPublicAddr(addr netip.Addr) bool {
|
|
if addr.Is4In6() {
|
|
addr = addr.Unmap()
|
|
}
|
|
return addr.IsGlobalUnicast() &&
|
|
!addr.IsPrivate() &&
|
|
!addr.IsLoopback() &&
|
|
!addr.IsLinkLocalUnicast() &&
|
|
!addr.IsLinkLocalMulticast() &&
|
|
!addr.IsMulticast() &&
|
|
!addr.IsUnspecified()
|
|
}
|