* feat(metadata): expand provider image cache queue * fix(metadata): harden provider image cache queue Addresses bug-review feedback from Codex/CodeRabbit on the metadata image cache pipeline. All findings validated against the code before fixing; false positives (rows/connection deadlock, PhotoSourcePath merge coupling) were confirmed non-issues and left unchanged. - Honor metadata.cache_images for the background processor. The cache_metadata_images task was registered whenever S3 was configured, so merely enabling object storage downloaded the entire provider-artwork catalog even with caching disabled. Add ImageCacheProcessor.SetEnabled, gate RunOnce/RunUntilIdle on it, and wire it (with hot reload) from cfg.Metadata.CacheImages in main.go. - Guard terminal job updates with lease ownership. EnqueueBatch can repurpose a running row with a new source; MarkSucceeded/MarkFailed keyed on id alone let a stale worker finalize the replacement job and drop the new artwork. Thread locked_by through and add status='running' AND locked_by=$n guards. - Avoid uploading stale jobs onto the live artwork key. Verify the target still references the job's source (CurrentTargetSourcePath) before CacheImage, so a job whose source an admin/refresh already replaced cannot overwrite the deterministic storage object. - COALESCE nullable external IDs in EnqueueExistingProviderArtwork. A NULL tmdb_id/tvdb_id/imdb_id on any candidate failed the scan and aborted the whole cache run; matches the existing item_repo pattern. - Stop re-downloading the catalog every 30 days. Discovery now skips targets whose *_path is already a cached relative path, making the cached row the durable dedup marker instead of the prunable job row. - Decouple catalog sweeps from queue draining. RunOnce no longer runs discovery per batch; RunUntilIdle sweeps only when the queue drains and throttles full sweeps to every 15m, so idle installs stop full-scanning every entity table each minute. - Requeue claimed-but-unstarted jobs on cancellation. Acquire the semaphore before spawning workers and RequeueClaimed any jobs not yet started, instead of leaving them locked until the 15m lease expires. - Skip the backoff sleep after the final upload attempt in putObjectWithRetry (saves ~1.5s on permanent failures). - Add the s3/file/local/upload/generated exclusion to the seasons and episodes backfill in migration 20260617184537 for consistency with the later migration (the bad backfill was inert downstream, but the asymmetry is removed). 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
79 lines
2.2 KiB
Go
79 lines
2.2 KiB
Go
package tasks
|
|
|
|
import (
|
|
"context"
|
|
"encoding/json"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/Silo-Server/silo-server/internal/metadata"
|
|
"github.com/Silo-Server/silo-server/internal/taskmanager"
|
|
)
|
|
|
|
type fakeMetadataImageCacheRunner struct {
|
|
stats metadata.ImageCacheRunStats
|
|
err error
|
|
claimLimit int
|
|
concurrency int
|
|
maxRuntime time.Duration
|
|
}
|
|
|
|
func (f *fakeMetadataImageCacheRunner) RunUntilIdle(_ context.Context, _ string, claimLimit int, concurrency int, maxRuntime time.Duration) (metadata.ImageCacheRunStats, error) {
|
|
f.claimLimit = claimLimit
|
|
f.concurrency = concurrency
|
|
f.maxRuntime = maxRuntime
|
|
return f.stats, f.err
|
|
}
|
|
|
|
type recordingProgress struct {
|
|
message string
|
|
}
|
|
|
|
func (r *recordingProgress) Report(_ float64, message string) {
|
|
r.message = message
|
|
}
|
|
|
|
func (r *recordingProgress) SetResultData(json.RawMessage) {}
|
|
|
|
func TestCacheMetadataImagesTaskProperties(t *testing.T) {
|
|
task := NewCacheMetadataImagesTask(&fakeMetadataImageCacheRunner{})
|
|
if task.Key() != "cache_metadata_images" {
|
|
t.Fatalf("Key() = %q", task.Key())
|
|
}
|
|
if task.Category() != taskmanager.TaskCategoryMetadata {
|
|
t.Fatalf("Category() = %q", task.Category())
|
|
}
|
|
if len(task.DefaultTriggers()) != 2 {
|
|
t.Fatalf("DefaultTriggers count = %d, want 2", len(task.DefaultTriggers()))
|
|
}
|
|
}
|
|
|
|
func TestCacheMetadataImagesTaskReportsStats(t *testing.T) {
|
|
runner := &fakeMetadataImageCacheRunner{
|
|
stats: metadata.ImageCacheRunStats{
|
|
Batches: 3,
|
|
EnqueuedExisting: 5,
|
|
Claimed: 4,
|
|
Succeeded: 3,
|
|
Failed: 1,
|
|
},
|
|
}
|
|
task := NewCacheMetadataImagesTask(runner)
|
|
progress := &recordingProgress{}
|
|
if err := task.Execute(context.Background(), progress); err != nil {
|
|
t.Fatalf("Execute() error = %v", err)
|
|
}
|
|
if runner.claimLimit != 1000 {
|
|
t.Fatalf("claimLimit = %d, want 1000", runner.claimLimit)
|
|
}
|
|
if runner.concurrency != 12 {
|
|
t.Fatalf("concurrency = %d, want 12", runner.concurrency)
|
|
}
|
|
if runner.maxRuntime != 10*time.Minute {
|
|
t.Fatalf("maxRuntime = %s, want 10m", runner.maxRuntime)
|
|
}
|
|
if progress.message != "Batches 3, enqueued 5 existing, claimed 4, cached 3, failed 1, skipped 0, deleted 0 old successes" {
|
|
t.Fatalf("progress message = %q", progress.message)
|
|
}
|
|
}
|