Files
CosmicClash/server/store/candidate_projection_test.go
T
Josh Creek 320ec46ba2 fix(server): partition and bound the Redis candidate projection
Both playlists shared one hash and sorted set, causing two independent
failures.

Starvation: Snapshot performed an unbounded ZRANGEBYSCORE and HMGET,
decoded the whole queue, and the matcher then truncated to its candidate
limit *before* filtering by playlist. A large casual prefix could
therefore leave the ranked worker with zero candidates indefinitely even
while ranked tickets were queued further down the set.

Mutual erasure: each matcher captured only its own playlist as the
durable source, but Rebuild replaced the shared keys, so a casual repair
wiped ranked projections and vice versa.

Namespace the keys per playlist, push the limit into Redis (LIMIT 0 N)
so reads no longer scale with total queue depth, and scope Rebuild to
one namespace. Rebuild now rejects a candidate whose playlist does not
match the namespace, which would reintroduce the starvation. Upsert
derives the namespace from the candidate; Remove takes the playlist,
since a ticket ID alone no longer identifies its namespace.

Add tests for a 300-deep casual backlog not starving ranked, for neither
playlist's rebuild erasing the other, and for the limit being applied
without losing enqueue ordering.
2026-09-05 10:23:52 +01:00

138 lines
5.5 KiB
Go

package store
import (
"context"
"testing"
"time"
"github.com/alicebob/miniredis/v2"
"github.com/cosmic-clash/cosmic-clash/server/domain"
"github.com/redis/go-redis/v9"
)
func TestCandidateProjectionRepairsPartialRedisStateFromDurableSource(t *testing.T) {
mini, err := miniredis.Run()
if err != nil {
t.Fatal(err)
}
defer mini.Close()
client := redis.NewClient(&redis.Options{Addr: mini.Addr()})
defer client.Close()
now := time.Unix(1000, 0).UTC()
candidate := domain.Candidate{Playlist: domain.Casual, TicketID: "repair-ticket", PlayerID: "repair-player", EnqueuedAt: now}
index := RedisCandidateIndex{Client: client, Prefix: "repair", TTL: time.Minute}
_, orderKey := index.keys(domain.Casual)
if err := client.ZAdd(context.Background(), orderKey, redis.Z{Score: float64(now.UnixNano()), Member: candidate.TicketID}).Err(); err != nil {
t.Fatal(err)
}
projection := CandidateProjection{Index: index, Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
return []domain.Candidate{candidate}, nil
}}
got, err := projection.Snapshot(context.Background(), domain.Casual, now, 1000)
if err != nil {
t.Fatal(err)
}
if len(got) != 1 || got[0].TicketID != candidate.TicketID {
t.Fatalf("repaired projection = %+v", got)
}
}
func TestCandidateProjectionDoesNotReturnCacheWhenRepairSourceFails(t *testing.T) {
index := RedisCandidateIndex{TTL: time.Minute}
projection := CandidateProjection{Index: index, Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
return nil, context.DeadlineExceeded
}}
if _, err := projection.Snapshot(context.Background(), domain.Casual, time.Unix(1000, 0), 1000); err == nil {
t.Fatal("cache projection succeeded without a usable Redis/index source")
}
}
// TestCandidateProjectionFallsBackToSourceWhenRedisIsEntirelyUnreachable
// covers the gap multiplayer-next.md §8.46 named "live Redis failover":
// Redis is documented everywhere (RedisCandidateIndex's own comment,
// cmd/matcher, cmd/control-plane) as an optional, rebuildable acceleration
// layer over PostgreSQL authority. Before this fix, Snapshot funnelled a
// genuine Redis connection failure into the same Repair path as an empty
// cache -- but Repair's own Index.Rebuild call also needs Redis, so it failed
// for the identical reason, and Snapshot returned an error even though the
// authoritative Source was perfectly healthy. A real Redis outage or
// mid-failover window would have taken matchmaking down completely.
func TestCandidateProjectionFallsBackToSourceWhenRedisIsEntirelyUnreachable(t *testing.T) {
mini, err := miniredis.Run()
if err != nil {
t.Fatal(err)
}
client := redis.NewClient(&redis.Options{Addr: mini.Addr()})
defer client.Close()
mini.Close() // Redis is now entirely unreachable, not merely empty or stale.
now := time.Unix(1000, 0).UTC()
candidate := domain.Candidate{Playlist: domain.Casual, TicketID: "down-ticket", PlayerID: "down-player", EnqueuedAt: now}
sourceCalls := 0
projection := CandidateProjection{
Index: RedisCandidateIndex{Client: client, Prefix: "down", TTL: time.Minute},
Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
sourceCalls++
return []domain.Candidate{candidate}, nil
},
}
got, err := projection.Snapshot(context.Background(), domain.Casual, now, 1000)
if err != nil {
t.Fatalf("Snapshot failed while Redis was down, even though Source (PostgreSQL) was healthy: %v", err)
}
if len(got) != 1 || got[0].TicketID != candidate.TicketID {
t.Fatalf("fallback snapshot = %+v, want the durable candidate served directly", got)
}
if sourceCalls != 1 {
t.Fatalf("Source calls = %d, want exactly 1", sourceCalls)
}
}
// TestCandidateProjectionStillFailsWhenBothRedisAndSourceAreDown proves the
// fallback isn't unconditional: if PostgreSQL itself is also unavailable,
// Snapshot must still fail rather than silently return an empty match pool.
func TestCandidateProjectionStillFailsWhenBothRedisAndSourceAreDown(t *testing.T) {
mini, err := miniredis.Run()
if err != nil {
t.Fatal(err)
}
client := redis.NewClient(&redis.Options{Addr: mini.Addr()})
defer client.Close()
mini.Close()
projection := CandidateProjection{
Index: RedisCandidateIndex{Client: client, Prefix: "down", TTL: time.Minute},
Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
return nil, context.DeadlineExceeded
},
}
if _, err := projection.Snapshot(context.Background(), domain.Casual, time.Unix(1000, 0), 1000); err == nil {
t.Fatal("Snapshot succeeded with both Redis and the durable source unavailable")
}
}
func TestCandidateProjectionRepairsEmptyIndexFromDurableSource(t *testing.T) {
mini, err := miniredis.Run()
if err != nil {
t.Fatal(err)
}
defer mini.Close()
client := redis.NewClient(&redis.Options{Addr: mini.Addr()})
defer client.Close()
now := time.Unix(1000, 0).UTC()
candidate := domain.Candidate{Playlist: domain.Casual, TicketID: "miss-ticket", PlayerID: "miss-player", EnqueuedAt: now}
projection := CandidateProjection{
Index: RedisCandidateIndex{Client: client, Prefix: "miss", TTL: time.Minute},
Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
return []domain.Candidate{candidate}, nil
},
}
got, err := projection.Snapshot(context.Background(), domain.Casual, now, 1000)
if err != nil {
t.Fatal(err)
}
if len(got) != 1 || got[0].TicketID != candidate.TicketID {
t.Fatalf("empty-index repair = %+v", got)
}
}