Five defects that all produced a WRONG over-cap count, which is why they land before the revocation batch: decision A1 raises the over-cap revocation TTL from 5m to ~24h, removing the self-healing that currently limits the damage of a miscount. A false positive after A1 blocks a legitimate stream for a day, so the count has to be trustworthy first. #1 -- overlapping edge requests deleted a live stream. Tracker.sessions was a set and Remove tore down all state plus the Redis key, while both proxy pour handlers deferred removal unconditionally. Two overlapping Range GETs on one session id -- ordinary seek behaviour -- meant the first to finish deleted the record while the second was still pouring, and later AddBytes calls were then dropped because AddBytes ignores bytes for a session with no live record. The stream went invisible to authoritative monitoring while still serving. Track now returns a Lease that the request-scoped caller releases exactly once; teardown happens when the last live lease is released. A plain refcount would have been wrong: Track(A) -> Remove -> Track(B) -> Release(A) decrements B, and clamping at zero does not help because the count legitimately belongs to B. That is not hypothetical -- the transcode node deliberately replaces sessions under the same id so a quality switch does not orphan ffmpeg, and it calls unconditional Remove from its reaper and stop paths. So each generation carries an epoch, Remove and Cleanup bump it, and a release from a superseded generation is a logged no-op. Lease identity is a set rather than a counter, which makes a duplicate release detectable instead of silently destructive. The transcode node keeps using Remove: its Track calls are not request-scoped and are correctly owned by session lifecycle. "Every Track needs a paired Release" is true only of the request-scoped callers. #8 -- async transcode tracking could leave a permanent ghost. The tracking write ran as a bare goroutine with a WithoutCancel context, so if stop won the race the delayed Track recreated the record after cleanup -- and because it landed in sessions, Snapshot treated it as live until Remove and it NEVER idle-expired. A permanent phantom inflating its owner's count, able to trigger false over-cap kills of that user's real streams. The write now takes the per-session lifecycle lock that stop and reap already hold, and re-checks session pointer identity before writing, so a stopped or replaced generation cannot resurrect a record. Pointer identity rather than id equality is what makes same-id replacement safe. The write stays off the request path -- the API server and the playback client are blocked on the 202. #9 + M3 -- protocol-v3 counted one stream twice. The stream token carries a transport id distinct from the logical session id, and the node tracked under the transport id while the API/proxy record used the logical one, so mergeStreams saw two streams. M3 was the reason this had not yet bitten: the v3 fresh-start caller sent no owner attribution at all, so the transport record landed under user 0, which the enforcer skips -- silently exempting the stream from the cap entirely. Fresh v3 starts now carry the logical session id and full owner attribution (both were already in scope at the call site), and merging is keyed on logical identity where present via one shared helper used by both merge functions, which had already drifted apart once. The enforcer view resolves SessionID to the logical id so a kill targets the real session rather than a replaceable transport generation. The raw admin view keeps the transport id and exposes logical_session_id as an additive omitempty field, advertised on the node-sessions capability endpoint, so the v1 response shape is unchanged. GAP-15 -- edge transcode liveness was request-observed. touchTranscodeSession fired before proxying, so hammering dead segment URLs advanced LastServedAt with zero bytes served. Visibility and liveness are now separate operations: EnsureEphemeral makes a session visible without claiming bytes were served, and served-byte liveness advances only from a 2xx/206 upstream response. Previously the proxy metered every upstream body regardless of status, so a node 404's error body counted as served bytes -- moving the touch later would not have fixed it. S4 -- LiveLocalSessions moved from the HTTP handlers package to streammonitor, which owns monitoring. A background enforcer importing api/handlers was backwards. Pure move; its existing mapping assertions moved with it. The LastActivityAt fallback inside it is left as-is -- decision A5 removes it in the liveness batch. Verified with go test -race across nodesessions, proxy and transcodenode; the overlap regression test was confirmed to fail under the old unconditional teardown. Part of #305.
382 lines
13 KiB
Go
382 lines
13 KiB
Go
package streammonitor
|
|
|
|
import (
|
|
"context"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/Silo-Server/silo-server/internal/nodesessions"
|
|
"github.com/Silo-Server/silo-server/internal/playback"
|
|
)
|
|
|
|
func fakeFn(infos []nodesessions.SessionInfo) func(ctx context.Context) ([]nodesessions.SessionInfo, error) {
|
|
return func(ctx context.Context) ([]nodesessions.SessionInfo, error) {
|
|
return infos, nil
|
|
}
|
|
}
|
|
|
|
func TestFuncSourceGrouping(t *testing.T) {
|
|
infos := []nodesessions.SessionInfo{
|
|
{SessionID: "s1", AuthUserID: 1, Type: "direct_play", StartedAt: "2026-07-04T10:00:00Z", LastServedAt: "2026-07-04T10:05:00Z"},
|
|
{SessionID: "s2", AuthUserID: 1, Type: "transcode", StartedAt: "2026-07-04T10:01:00Z", LastServedAt: "2026-07-04T10:06:00Z"},
|
|
{SessionID: "s3", AuthUserID: 2, Type: "remux", StartedAt: "2026-07-04T10:02:00Z", LastServedAt: "2026-07-04T10:07:00Z"},
|
|
{SessionID: "s4", AuthUserID: 0, Type: "direct_play"}, // no user, no timestamps
|
|
}
|
|
|
|
snap, err := NewFuncSource(fakeFn(infos)).Snapshot(context.Background())
|
|
if err != nil {
|
|
t.Fatalf("Snapshot: %v", err)
|
|
}
|
|
|
|
if got := len(snap.Streams); got != 4 {
|
|
t.Fatalf("Streams len = %d, want 4", got)
|
|
}
|
|
|
|
if got := snap.CountByUser(1); got != 2 {
|
|
t.Errorf("CountByUser(1) = %d, want 2", got)
|
|
}
|
|
if got := snap.CountByUser(2); got != 1 {
|
|
t.Errorf("CountByUser(2) = %d, want 1", got)
|
|
}
|
|
if got := snap.CountByUser(0); got != 1 {
|
|
t.Errorf("CountByUser(0) = %d, want 1", got)
|
|
}
|
|
if got := snap.CountByUser(99); got != 0 {
|
|
t.Errorf("CountByUser(99) = %d, want 0", got)
|
|
}
|
|
|
|
u1 := snap.StreamsForUser(1)
|
|
if len(u1) != 2 {
|
|
t.Fatalf("StreamsForUser(1) len = %d, want 2", len(u1))
|
|
}
|
|
for _, st := range u1 {
|
|
if st.UserID != 1 {
|
|
t.Errorf("StreamsForUser(1) returned stream with UserID %d", st.UserID)
|
|
}
|
|
}
|
|
|
|
byUser := snap.ByUser()
|
|
if len(byUser) != 3 {
|
|
t.Fatalf("ByUser len = %d, want 3 (users 0,1,2)", len(byUser))
|
|
}
|
|
if len(byUser[1]) != 2 || len(byUser[2]) != 1 || len(byUser[0]) != 1 {
|
|
t.Errorf("ByUser grouping wrong: %#v", map[int]int{0: len(byUser[0]), 1: len(byUser[1]), 2: len(byUser[2])})
|
|
}
|
|
}
|
|
|
|
func TestTimestampParsing(t *testing.T) {
|
|
infos := []nodesessions.SessionInfo{
|
|
{SessionID: "s1", AuthUserID: 1, StartedAt: "2026-07-04T10:00:00Z", LastServedAt: "2026-07-04T10:05:00Z"},
|
|
{SessionID: "s2", AuthUserID: 1, StartedAt: "not-a-time", LastServedAt: ""},
|
|
}
|
|
snap, err := NewFuncSource(fakeFn(infos)).Snapshot(context.Background())
|
|
if err != nil {
|
|
t.Fatalf("Snapshot: %v", err)
|
|
}
|
|
bySession := map[string]LiveStream{}
|
|
for _, st := range snap.Streams {
|
|
bySession[st.SessionID] = st
|
|
}
|
|
|
|
want := time.Date(2026, 7, 4, 10, 5, 0, 0, time.UTC)
|
|
if !bySession["s1"].LastServedAt.Equal(want) {
|
|
t.Errorf("s1 LastServedAt = %v, want %v", bySession["s1"].LastServedAt, want)
|
|
}
|
|
if !bySession["s2"].StartedAt.IsZero() {
|
|
t.Errorf("s2 StartedAt = %v, want zero (unparseable)", bySession["s2"].StartedAt)
|
|
}
|
|
if !bySession["s2"].LastServedAt.IsZero() {
|
|
t.Errorf("s2 LastServedAt = %v, want zero (empty)", bySession["s2"].LastServedAt)
|
|
}
|
|
}
|
|
|
|
func TestDedupeKeepsNewest(t *testing.T) {
|
|
// Same SessionID observed on two nodes (proxy vs transcode node). Keep the
|
|
// record with the most recent LastServedAt.
|
|
infos := []nodesessions.SessionInfo{
|
|
{SessionID: "dup", AuthUserID: 5, NodeName: "proxy", LastServedAt: "2026-07-04T10:00:00Z", BytesServed: 100},
|
|
{SessionID: "dup", AuthUserID: 5, NodeName: "transcode", LastServedAt: "2026-07-04T10:09:00Z", BytesServed: 900},
|
|
{SessionID: "other", AuthUserID: 5, NodeName: "proxy", LastServedAt: "2026-07-04T10:03:00Z"},
|
|
}
|
|
snap, err := NewFuncSource(fakeFn(infos)).Snapshot(context.Background())
|
|
if err != nil {
|
|
t.Fatalf("Snapshot: %v", err)
|
|
}
|
|
|
|
if got := len(snap.Streams); got != 2 {
|
|
t.Fatalf("Streams len = %d, want 2 (deduped)", got)
|
|
}
|
|
if got := snap.CountByUser(5); got != 2 {
|
|
t.Errorf("CountByUser(5) = %d, want 2", got)
|
|
}
|
|
|
|
var dup LiveStream
|
|
found := false
|
|
for _, st := range snap.Streams {
|
|
if st.SessionID == "dup" {
|
|
dup = st
|
|
found = true
|
|
}
|
|
}
|
|
if !found {
|
|
t.Fatal("deduped session 'dup' not present")
|
|
}
|
|
if dup.NodeName != "transcode" {
|
|
t.Errorf("dedupe kept NodeName %q, want transcode (newest LastServedAt)", dup.NodeName)
|
|
}
|
|
if dup.BytesServed != 900 {
|
|
t.Errorf("dedupe kept BytesServed %d, want 900", dup.BytesServed)
|
|
}
|
|
}
|
|
|
|
func TestMergeStreamsDirect(t *testing.T) {
|
|
// Direct unit test of the merge helper, mirroring RedisSource dedupe.
|
|
in := []LiveStream{
|
|
{SessionID: "a", NodeName: "n1", LastServedAt: time.Unix(100, 0)},
|
|
{SessionID: "a", NodeName: "n2", LastServedAt: time.Unix(200, 0)},
|
|
{SessionID: "b", NodeName: "n1", LastServedAt: time.Unix(150, 0)},
|
|
}
|
|
out := mergeStreams(in)
|
|
if len(out) != 2 {
|
|
t.Fatalf("mergeStreams len = %d, want 2", len(out))
|
|
}
|
|
for _, st := range out {
|
|
if st.SessionID == "a" && st.NodeName != "n2" {
|
|
t.Errorf("merge kept %q for session a, want n2 (newest)", st.NodeName)
|
|
}
|
|
}
|
|
}
|
|
|
|
func TestMergeStreamsCarriesOwnershipForward(t *testing.T) {
|
|
// The transcode node's own start record has no resolved owner (UserID 0) and
|
|
// can be the freshest copy of a session. Taking it wholesale would bucket the
|
|
// session under user 0, which the enforcer skips — exempting it from the cap.
|
|
// The merge must recover the owner from the proxy's (staler) owned record.
|
|
in := []LiveStream{
|
|
{SessionID: "s", NodeName: "proxy", UserID: 42, ProfileID: "p1", MediaFileID: 7, LastServedAt: time.Unix(100, 0)},
|
|
{SessionID: "s", NodeName: "transcode", UserID: 0, LastServedAt: time.Unix(200, 0)},
|
|
}
|
|
out := mergeStreams(in)
|
|
if len(out) != 1 {
|
|
t.Fatalf("mergeStreams len = %d, want 1", len(out))
|
|
}
|
|
got := out[0]
|
|
if got.NodeName != "transcode" {
|
|
t.Errorf("kept NodeName %q, want transcode (freshest)", got.NodeName)
|
|
}
|
|
if got.UserID != 42 {
|
|
t.Errorf("UserID = %d, want 42 (recovered from the proxy record)", got.UserID)
|
|
}
|
|
if got.ProfileID != "p1" || got.MediaFileID != 7 {
|
|
t.Errorf("ownership not fully recovered: ProfileID=%q MediaFileID=%d", got.ProfileID, got.MediaFileID)
|
|
}
|
|
}
|
|
|
|
func TestToLiveStreamCarriesRouteAndClient(t *testing.T) {
|
|
// The normalized view must surface route (native vs jellycompat) and client
|
|
// identity so the monitor is first-class, not just session id + method.
|
|
info := nodesessions.SessionInfo{
|
|
SessionID: "s1",
|
|
AuthUserID: 9,
|
|
Type: "transcode",
|
|
Route: "jellycompat",
|
|
ClientIP: "203.0.113.7",
|
|
ClientName: "Infuse",
|
|
Position: 61.5,
|
|
HWAccel: "vaapi",
|
|
}
|
|
got := toLiveStream(info)
|
|
if got.Route != "jellycompat" || got.ClientIP != "203.0.113.7" || got.ClientName != "Infuse" {
|
|
t.Fatalf("route/client not carried: %+v", got)
|
|
}
|
|
if got.Position != 61.5 || got.HWAccel != "vaapi" {
|
|
t.Fatalf("position/hwaccel not carried: %+v", got)
|
|
}
|
|
}
|
|
|
|
func TestMergeStreamsBackfillsAttribution(t *testing.T) {
|
|
// A fresher-but-thinner record (e.g. an ownerless transcode-node start record)
|
|
// must not drop route/client that a staler record for the same session carries.
|
|
in := []LiveStream{
|
|
{SessionID: "s", NodeName: "proxy", UserID: 5, Route: "native", ClientName: "SiloTV", ClientIP: "10.0.0.9", LastServedAt: time.Unix(100, 0)},
|
|
{SessionID: "s", NodeName: "transcode", UserID: 0, LastServedAt: time.Unix(200, 0)},
|
|
}
|
|
out := mergeStreams(in)
|
|
if len(out) != 1 {
|
|
t.Fatalf("merge len = %d, want 1", len(out))
|
|
}
|
|
got := out[0]
|
|
if got.NodeName != "transcode" {
|
|
t.Errorf("kept NodeName %q, want transcode (freshest)", got.NodeName)
|
|
}
|
|
if got.UserID != 5 || got.Route != "native" || got.ClientName != "SiloTV" || got.ClientIP != "10.0.0.9" {
|
|
t.Errorf("attribution not backfilled from staler record: %+v", got)
|
|
}
|
|
}
|
|
|
|
func TestMergeStreamsKeepsLargestObservedByteTotal(t *testing.T) {
|
|
for _, tc := range []struct {
|
|
name string
|
|
older, newer int64
|
|
want int64
|
|
}{
|
|
{name: "freshest has zero", older: 8192, newer: 0, want: 8192},
|
|
{name: "stale has zero", older: 0, newer: 8192, want: 8192},
|
|
} {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
out := mergeStreams([]LiveStream{
|
|
{SessionID: "s", BytesServed: tc.older, LastServedAt: time.Unix(100, 0)},
|
|
{SessionID: "s", BytesServed: tc.newer, LastServedAt: time.Unix(200, 0)},
|
|
})
|
|
if len(out) != 1 || out[0].BytesServed != tc.want {
|
|
t.Fatalf("merge = %+v, want bytes %d", out, tc.want)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
func TestFuncSourceNilFn(t *testing.T) {
|
|
snap, err := NewFuncSource(nil).Snapshot(context.Background())
|
|
if err != nil {
|
|
t.Fatalf("Snapshot: %v", err)
|
|
}
|
|
if len(snap.Streams) != 0 {
|
|
t.Errorf("nil fn Streams len = %d, want 0", len(snap.Streams))
|
|
}
|
|
}
|
|
|
|
func TestRedisSourceNilClient(t *testing.T) {
|
|
snap, err := NewRedisSource(nil).Snapshot(context.Background())
|
|
if err != nil {
|
|
t.Fatalf("Snapshot: %v", err)
|
|
}
|
|
if len(snap.Streams) != 0 {
|
|
t.Errorf("nil client Streams len = %d, want 0", len(snap.Streams))
|
|
}
|
|
}
|
|
|
|
// TestDedupeSessionInfos mirrors the mergeStreams rules on the raw SessionInfo
|
|
// shape used by the admin session list: one row per session id, freshest copy
|
|
// wins, a resolved owner and missing attribution are carried across the merge
|
|
// so the operator view matches the enforcer's picture (no double-counted
|
|
// streams, no user-0 rows when any copy knows the owner).
|
|
func TestDedupeSessionInfos(t *testing.T) {
|
|
newer := time.Now().UTC().Format(time.RFC3339)
|
|
older := time.Now().Add(-time.Minute).UTC().Format(time.RFC3339)
|
|
infos := []nodesessions.SessionInfo{
|
|
{SessionID: "s1", NodeName: "central", AuthUserID: 7, ProfileID: "p1", MediaFileID: 42,
|
|
Route: "native", ClientName: "SiloTV", Position: 130, LastServedAt: older},
|
|
// The same stream, seen from the edge serving it: freshest but ownerless.
|
|
{SessionID: "s1", NodeName: "edge-1", LastServedAt: newer, BytesServed: 9000},
|
|
{SessionID: "s2", NodeName: "edge-1", AuthUserID: 8, LastServedAt: newer},
|
|
}
|
|
|
|
out := DedupeSessionInfos(infos)
|
|
if len(out) != 2 {
|
|
t.Fatalf("dedupe len = %d, want 2", len(out))
|
|
}
|
|
byID := make(map[string]nodesessions.SessionInfo, len(out))
|
|
for _, info := range out {
|
|
byID[info.SessionID] = info
|
|
}
|
|
s1, ok := byID["s1"]
|
|
if !ok {
|
|
t.Fatalf("s1 missing from dedupe output")
|
|
}
|
|
if s1.NodeName != "edge-1" || s1.BytesServed != 9000 {
|
|
t.Errorf("s1 freshest copy not kept: %+v", s1)
|
|
}
|
|
if s1.AuthUserID != 7 || s1.ProfileID != "p1" || s1.MediaFileID != 42 {
|
|
t.Errorf("s1 owner not carried across merge: %+v", s1)
|
|
}
|
|
if s1.Route != "native" || s1.ClientName != "SiloTV" || s1.Position != 130 {
|
|
t.Errorf("s1 attribution not backfilled: %+v", s1)
|
|
}
|
|
if s2 := byID["s2"]; s2.AuthUserID != 8 {
|
|
t.Errorf("s2 mangled by dedupe: %+v", s2)
|
|
}
|
|
}
|
|
|
|
func TestDedupeSessionInfosKeepsLargestObservedByteTotal(t *testing.T) {
|
|
older := time.Now().Add(-time.Minute).UTC().Format(time.RFC3339Nano)
|
|
newer := time.Now().UTC().Format(time.RFC3339Nano)
|
|
for _, tc := range []struct {
|
|
name string
|
|
older, newer int64
|
|
want int64
|
|
}{
|
|
{name: "freshest has zero", older: 4096, newer: 0, want: 4096},
|
|
{name: "stale has zero", older: 0, newer: 4096, want: 4096},
|
|
} {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
out := DedupeSessionInfos([]nodesessions.SessionInfo{
|
|
{SessionID: "s", BytesServed: tc.older, LastServedAt: older},
|
|
{SessionID: "s", BytesServed: tc.newer, LastServedAt: newer},
|
|
})
|
|
if len(out) != 1 || out[0].BytesServed != tc.want {
|
|
t.Fatalf("dedupe = %+v, want bytes %d", out, tc.want)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
func TestLogicalAndTransportRecordsMergeWithOwnerResolved(t *testing.T) {
|
|
out := mergeStreams([]LiveStream{
|
|
{SessionID: "logical", UserID: 17, ProfileID: "p", LastServedAt: time.Unix(100, 0)},
|
|
{SessionID: "transport-a", LogicalSessionID: "logical", LastServedAt: time.Unix(200, 0)},
|
|
})
|
|
if len(out) != 1 {
|
|
t.Fatalf("merge len = %d, want 1", len(out))
|
|
}
|
|
if out[0].SessionID != "logical" || out[0].UserID != 17 || out[0].ProfileID != "p" {
|
|
t.Fatalf("canonical merged stream = %+v", out[0])
|
|
}
|
|
}
|
|
|
|
func TestDistinctLogicalStreamsStayDistinct(t *testing.T) {
|
|
out := mergeStreams([]LiveStream{
|
|
{SessionID: "transport-a", LogicalSessionID: "logical-a"},
|
|
{SessionID: "transport-b", LogicalSessionID: "logical-b"},
|
|
})
|
|
if len(out) != 2 {
|
|
t.Fatalf("distinct logical streams merged: %+v", out)
|
|
}
|
|
}
|
|
|
|
func TestTransportGenerationsOfOneLogicalSessionMerge(t *testing.T) {
|
|
out := DedupeSessionInfos([]nodesessions.SessionInfo{
|
|
{SessionID: "transport-a", LogicalSessionID: "logical", NodeName: "old", LastServedAt: "2026-07-30T01:00:00Z"},
|
|
{SessionID: "transport-b", LogicalSessionID: "logical", NodeName: "new", LastServedAt: "2026-07-30T02:00:00Z"},
|
|
})
|
|
if len(out) != 1 || out[0].SessionID != "transport-b" || out[0].LogicalSessionID != "logical" {
|
|
t.Fatalf("transport generation dedupe = %+v", out)
|
|
}
|
|
}
|
|
|
|
func TestLiveLocalSessionsMapping(t *testing.T) {
|
|
sm := playback.NewSessionManager(0, 0)
|
|
ctx := playback.WithClientInfo(context.Background(), playback.ClientInfo{
|
|
Name: "Silo TV",
|
|
})
|
|
session, err := sm.StartSessionWithContext(ctx, 42, "profile-1", 9, playback.PlayDirect, false)
|
|
if err != nil {
|
|
t.Fatalf("StartSessionWithContext: %v", err)
|
|
}
|
|
session.ClientIP = "192.0.2.10"
|
|
got := LiveLocalSessions(sm, "local")
|
|
if len(got) != 1 {
|
|
t.Fatalf("sessions = %+v", got)
|
|
}
|
|
info := got[0]
|
|
if info.SessionID != session.ID || info.NodeName != "local" ||
|
|
info.AuthUserID != 42 || info.ProfileID != "profile-1" ||
|
|
info.MediaFileID != 9 || info.Type != string(playback.PlayDirect) ||
|
|
info.Route != session.Origin() || info.ClientName != "Silo TV" ||
|
|
info.ClientIP != "192.0.2.10" {
|
|
t.Fatalf("mapped session = %+v", info)
|
|
}
|
|
if info.LastServedAt != session.LastActivityAt.UTC().Format(time.RFC3339) {
|
|
t.Fatalf("LastServedAt = %q, want LastActivityAt fallback", info.LastServedAt)
|
|
}
|
|
}
|