Files
silo-server/internal/taskmanager/tasks/cache_metadata_images_test.go
T
14ffc91dfb [codex] Expand provider image cache queue (#176)
* feat(metadata): expand provider image cache queue

* fix(metadata): harden provider image cache queue

Addresses bug-review feedback from Codex/CodeRabbit on the metadata image
cache pipeline. All findings validated against the code before fixing;
false positives (rows/connection deadlock, PhotoSourcePath merge coupling)
were confirmed non-issues and left unchanged.

- Honor metadata.cache_images for the background processor. The
  cache_metadata_images task was registered whenever S3 was configured,
  so merely enabling object storage downloaded the entire provider-artwork
  catalog even with caching disabled. Add ImageCacheProcessor.SetEnabled,
  gate RunOnce/RunUntilIdle on it, and wire it (with hot reload) from
  cfg.Metadata.CacheImages in main.go.
- Guard terminal job updates with lease ownership. EnqueueBatch can
  repurpose a running row with a new source; MarkSucceeded/MarkFailed
  keyed on id alone let a stale worker finalize the replacement job and
  drop the new artwork. Thread locked_by through and add
  status='running' AND locked_by=$n guards.
- Avoid uploading stale jobs onto the live artwork key. Verify the
  target still references the job's source (CurrentTargetSourcePath)
  before CacheImage, so a job whose source an admin/refresh already
  replaced cannot overwrite the deterministic storage object.
- COALESCE nullable external IDs in EnqueueExistingProviderArtwork. A
  NULL tmdb_id/tvdb_id/imdb_id on any candidate failed the scan and
  aborted the whole cache run; matches the existing item_repo pattern.
- Stop re-downloading the catalog every 30 days. Discovery now skips
  targets whose *_path is already a cached relative path, making the
  cached row the durable dedup marker instead of the prunable job row.
- Decouple catalog sweeps from queue draining. RunOnce no longer runs
  discovery per batch; RunUntilIdle sweeps only when the queue drains and
  throttles full sweeps to every 15m, so idle installs stop full-scanning
  every entity table each minute.
- Requeue claimed-but-unstarted jobs on cancellation. Acquire the
  semaphore before spawning workers and RequeueClaimed any jobs not yet
  started, instead of leaving them locked until the 15m lease expires.
- Skip the backoff sleep after the final upload attempt in
  putObjectWithRetry (saves ~1.5s on permanent failures).
- Add the s3/file/local/upload/generated exclusion to the seasons and
  episodes backfill in migration 20260617184537 for consistency with the
  later migration (the bad backfill was inert downstream, but the
  asymmetry is removed).

🤖 Generated with [Claude Code](https://claude.com/claude-code)

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-18 10:07:58 -04:00

79 lines
2.2 KiB
Go

package tasks
import (
"context"
"encoding/json"
"testing"
"time"
"github.com/Silo-Server/silo-server/internal/metadata"
"github.com/Silo-Server/silo-server/internal/taskmanager"
)
type fakeMetadataImageCacheRunner struct {
stats metadata.ImageCacheRunStats
err error
claimLimit int
concurrency int
maxRuntime time.Duration
}
func (f *fakeMetadataImageCacheRunner) RunUntilIdle(_ context.Context, _ string, claimLimit int, concurrency int, maxRuntime time.Duration) (metadata.ImageCacheRunStats, error) {
f.claimLimit = claimLimit
f.concurrency = concurrency
f.maxRuntime = maxRuntime
return f.stats, f.err
}
type recordingProgress struct {
message string
}
func (r *recordingProgress) Report(_ float64, message string) {
r.message = message
}
func (r *recordingProgress) SetResultData(json.RawMessage) {}
func TestCacheMetadataImagesTaskProperties(t *testing.T) {
task := NewCacheMetadataImagesTask(&fakeMetadataImageCacheRunner{})
if task.Key() != "cache_metadata_images" {
t.Fatalf("Key() = %q", task.Key())
}
if task.Category() != taskmanager.TaskCategoryMetadata {
t.Fatalf("Category() = %q", task.Category())
}
if len(task.DefaultTriggers()) != 2 {
t.Fatalf("DefaultTriggers count = %d, want 2", len(task.DefaultTriggers()))
}
}
func TestCacheMetadataImagesTaskReportsStats(t *testing.T) {
runner := &fakeMetadataImageCacheRunner{
stats: metadata.ImageCacheRunStats{
Batches: 3,
EnqueuedExisting: 5,
Claimed: 4,
Succeeded: 3,
Failed: 1,
},
}
task := NewCacheMetadataImagesTask(runner)
progress := &recordingProgress{}
if err := task.Execute(context.Background(), progress); err != nil {
t.Fatalf("Execute() error = %v", err)
}
if runner.claimLimit != 1000 {
t.Fatalf("claimLimit = %d, want 1000", runner.claimLimit)
}
if runner.concurrency != 12 {
t.Fatalf("concurrency = %d, want 12", runner.concurrency)
}
if runner.maxRuntime != 10*time.Minute {
t.Fatalf("maxRuntime = %s, want 10m", runner.maxRuntime)
}
if progress.message != "Batches 3, enqueued 5 existing, claimed 4, cached 3, failed 1, skipped 0, deleted 0 old successes" {
t.Fatalf("progress message = %q", progress.message)
}
}