Files
silo-server/internal/playback/transcode_restart_guard_test.go
T
a4402d4621 fix(playback): stop seek-restart churn and premature paused-session reaping (#279)
* fix(playback): stop seek-restart churn and premature paused-session reaping

Two latent defects (present since the initial migration) explain the
cross-client freezes on resume-after-pause, intro-skip, and manual seek
reported in #243:

1. Dueling transcode restarts. TranscodeSession.Restart had no
   concurrency guard, and SegmentRecoveryDecision told concurrent
   segment requests to restart again while a restart was already in
   flight (the restart window runs with running=false, so requests hit
   the restart-happy transcode_not_running branch before the Restarting
   check). Pipelined HLS segment requests therefore spawned restarts
   that kept killing the ffmpeg the player was waiting on — a 30s stall
   and another restart, surfacing as buffer-then-freeze. Restart is now
   single-flight and the recovery decision waits out an in-flight
   restart (WaitForSegment already polls politely through one).

2. Paused sessions were reaped after 2 minutes, killing the transcode
   with no revival path — every resume request then 404s. Clients that
   stop progress reporting while paused (tvOS; backgrounded web tabs)
   froze on play after a >5 minute pause. Paused grace is now 30
   minutes in both the in-memory session manager and the DB session
   cleaner.

Also unifies post-restart re-arming: the throttler and exit monitor are
re-armed via a restart hook installed at session creation and fired by
Restart itself, so the jellycompat seek path (which previously re-armed
nothing) and the audio-switch path (which re-armed only the throttler)
now behave identically to web segment recovery.

Follow-up (not in this PR): lazily revive a reaped session's transcode
on resume instead of 404ing, so even multi-hour pauses recover.

Fixes #243

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

* docs(playback): scope the restart-hook comment to handler-created sessions

The comment above SetRestartHook claimed the hook covers restarts from
'jellycompat seek', but jellycompat creates its own TranscodeSessions via
playback.StartTranscode in a separate registry and calls Restart directly;
those sessions never get the hook and never had throttler/exit-monitor
wiring to re-arm. Reword the comment (and the matching test comment) to
scope the guarantee to handler-created sessions.

Also fix TestRestartInvokesRestartHook to resolve `true` via PATH instead
of hardcoding /bin/true, which does not exist on macOS.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
Co-authored-by: Quick <31828688+Quick104@users.noreply.github.com>
2026-07-02 14:06:25 -04:00

112 lines
3.5 KiB
Go

package playback
import (
"context"
"os/exec"
"testing"
"time"
)
// TestSegmentRecoveryDecisionWaitsWhileRestarting covers half of issue #243's
// seek-freeze: while a restart is already in flight, a concurrent segment
// request must WAIT for the restart's output rather than trigger another
// restart. Without this, pipelined HLS segment requests spawn dueling ffmpeg
// restarts that keep preempting the segment the player is blocked on.
func TestSegmentRecoveryDecisionWaitsWhileRestarting(t *testing.T) {
session := &TranscodeSession{
outputDir: t.TempDir(),
restarting: true,
opts: TranscodeOpts{
TargetCodecVideo: "h264",
SegmentDuration: 2,
StartSegmentNumber: 0,
},
}
decision := session.SegmentRecoveryDecision(10, time.Now())
if decision.Reason != "transcode_restarting" {
t.Fatalf("Reason = %q, want transcode_restarting", decision.Reason)
}
if !decision.Wait {
t.Error("Wait = false, want true (concurrent requests must wait out an in-flight restart)")
}
if decision.RestartOnTimeout {
t.Error("RestartOnTimeout = true, want false (a timed-out wait must re-decide, not blindly restart)")
}
}
// TestRestartInvokesRestartHook verifies that a successful Restart fires the
// session's restart hook. The API handler uses the hook to re-arm the
// throttler and the exit monitor; firing it from Restart itself keeps every
// restart caller of a hook-wired session (web segment recovery, audio
// switch) consistent instead of each call site remembering to re-arm by
// hand.
func TestRestartInvokesRestartHook(t *testing.T) {
// `true` starts and exits cleanly, standing in for ffmpeg. Resolve it
// via PATH — it lives in /bin on Linux but /usr/bin on macOS.
truePath, err := exec.LookPath("true")
if err != nil {
t.Skipf("`true` not found in PATH: %v", err)
}
session := &TranscodeSession{
outputDir: t.TempDir(),
opts: TranscodeOpts{
TargetCodecVideo: "h264",
SegmentDuration: 2,
StartSegmentNumber: 0,
FFmpegPath: truePath,
},
}
hookFired := make(chan struct{}, 1)
session.SetRestartHook(func(context.Context) {
hookFired <- struct{}{}
})
if err := session.Restart(context.Background(), 20, 10); err != nil {
t.Fatalf("Restart: %v", err)
}
select {
case <-hookFired:
case <-time.After(2 * time.Second):
t.Fatal("restart hook was not invoked after successful restart")
}
}
// TestRestartIsSingleFlight covers the other half: Restart must be
// single-flight per session. A second caller arriving while a restart is in
// progress must return immediately without killing the process the first
// restart just started.
func TestRestartIsSingleFlight(t *testing.T) {
session := &TranscodeSession{
outputDir: t.TempDir(),
restarting: true,
opts: TranscodeOpts{
TargetCodecVideo: "h264",
SegmentDuration: 2,
StartSegmentNumber: 0,
// Nonexistent binary: if the guard is missing and Restart
// proceeds, exec fails and the call returns an error, failing
// the assertions below.
FFmpegPath: "/nonexistent/ffmpeg-single-flight-test",
},
}
err := session.Restart(context.Background(), 20, 10)
if err != nil {
t.Fatalf("Restart during in-flight restart = %v, want nil (single-flight no-op)", err)
}
session.mu.Lock()
restartCount := session.restartCount
stillRestarting := session.restarting
session.mu.Unlock()
if restartCount != 0 {
t.Errorf("restartCount = %d, want 0 (second caller must not perform a restart)", restartCount)
}
if !stillRestarting {
t.Error("restarting flag cleared by no-op caller; must be left for the in-flight restart to clear")
}
}