mirror of
https://github.com/jcreek/CosmicClash.git
synced 2026-09-10 16:04:04 +00:00
320ec46ba2
Both playlists shared one hash and sorted set, causing two independent failures. Starvation: Snapshot performed an unbounded ZRANGEBYSCORE and HMGET, decoded the whole queue, and the matcher then truncated to its candidate limit *before* filtering by playlist. A large casual prefix could therefore leave the ranked worker with zero candidates indefinitely even while ranked tickets were queued further down the set. Mutual erasure: each matcher captured only its own playlist as the durable source, but Rebuild replaced the shared keys, so a casual repair wiped ranked projections and vice versa. Namespace the keys per playlist, push the limit into Redis (LIMIT 0 N) so reads no longer scale with total queue depth, and scope Rebuild to one namespace. Rebuild now rejects a candidate whose playlist does not match the namespace, which would reintroduce the starvation. Upsert derives the namespace from the candidate; Remove takes the playlist, since a ticket ID alone no longer identifies its namespace. Add tests for a 300-deep casual backlog not starving ranked, for neither playlist's rebuild erasing the other, and for the limit being applied without losing enqueue ordering.
138 lines
5.5 KiB
Go
138 lines
5.5 KiB
Go
package store
|
|
|
|
import (
|
|
"context"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/alicebob/miniredis/v2"
|
|
"github.com/cosmic-clash/cosmic-clash/server/domain"
|
|
"github.com/redis/go-redis/v9"
|
|
)
|
|
|
|
func TestCandidateProjectionRepairsPartialRedisStateFromDurableSource(t *testing.T) {
|
|
mini, err := miniredis.Run()
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
defer mini.Close()
|
|
client := redis.NewClient(&redis.Options{Addr: mini.Addr()})
|
|
defer client.Close()
|
|
now := time.Unix(1000, 0).UTC()
|
|
candidate := domain.Candidate{Playlist: domain.Casual, TicketID: "repair-ticket", PlayerID: "repair-player", EnqueuedAt: now}
|
|
index := RedisCandidateIndex{Client: client, Prefix: "repair", TTL: time.Minute}
|
|
_, orderKey := index.keys(domain.Casual)
|
|
if err := client.ZAdd(context.Background(), orderKey, redis.Z{Score: float64(now.UnixNano()), Member: candidate.TicketID}).Err(); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
projection := CandidateProjection{Index: index, Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
|
|
return []domain.Candidate{candidate}, nil
|
|
}}
|
|
got, err := projection.Snapshot(context.Background(), domain.Casual, now, 1000)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(got) != 1 || got[0].TicketID != candidate.TicketID {
|
|
t.Fatalf("repaired projection = %+v", got)
|
|
}
|
|
}
|
|
|
|
func TestCandidateProjectionDoesNotReturnCacheWhenRepairSourceFails(t *testing.T) {
|
|
index := RedisCandidateIndex{TTL: time.Minute}
|
|
projection := CandidateProjection{Index: index, Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
|
|
return nil, context.DeadlineExceeded
|
|
}}
|
|
if _, err := projection.Snapshot(context.Background(), domain.Casual, time.Unix(1000, 0), 1000); err == nil {
|
|
t.Fatal("cache projection succeeded without a usable Redis/index source")
|
|
}
|
|
}
|
|
|
|
// TestCandidateProjectionFallsBackToSourceWhenRedisIsEntirelyUnreachable
|
|
// covers the gap multiplayer-next.md §8.46 named "live Redis failover":
|
|
// Redis is documented everywhere (RedisCandidateIndex's own comment,
|
|
// cmd/matcher, cmd/control-plane) as an optional, rebuildable acceleration
|
|
// layer over PostgreSQL authority. Before this fix, Snapshot funnelled a
|
|
// genuine Redis connection failure into the same Repair path as an empty
|
|
// cache -- but Repair's own Index.Rebuild call also needs Redis, so it failed
|
|
// for the identical reason, and Snapshot returned an error even though the
|
|
// authoritative Source was perfectly healthy. A real Redis outage or
|
|
// mid-failover window would have taken matchmaking down completely.
|
|
func TestCandidateProjectionFallsBackToSourceWhenRedisIsEntirelyUnreachable(t *testing.T) {
|
|
mini, err := miniredis.Run()
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
client := redis.NewClient(&redis.Options{Addr: mini.Addr()})
|
|
defer client.Close()
|
|
mini.Close() // Redis is now entirely unreachable, not merely empty or stale.
|
|
|
|
now := time.Unix(1000, 0).UTC()
|
|
candidate := domain.Candidate{Playlist: domain.Casual, TicketID: "down-ticket", PlayerID: "down-player", EnqueuedAt: now}
|
|
sourceCalls := 0
|
|
projection := CandidateProjection{
|
|
Index: RedisCandidateIndex{Client: client, Prefix: "down", TTL: time.Minute},
|
|
Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
|
|
sourceCalls++
|
|
return []domain.Candidate{candidate}, nil
|
|
},
|
|
}
|
|
got, err := projection.Snapshot(context.Background(), domain.Casual, now, 1000)
|
|
if err != nil {
|
|
t.Fatalf("Snapshot failed while Redis was down, even though Source (PostgreSQL) was healthy: %v", err)
|
|
}
|
|
if len(got) != 1 || got[0].TicketID != candidate.TicketID {
|
|
t.Fatalf("fallback snapshot = %+v, want the durable candidate served directly", got)
|
|
}
|
|
if sourceCalls != 1 {
|
|
t.Fatalf("Source calls = %d, want exactly 1", sourceCalls)
|
|
}
|
|
}
|
|
|
|
// TestCandidateProjectionStillFailsWhenBothRedisAndSourceAreDown proves the
|
|
// fallback isn't unconditional: if PostgreSQL itself is also unavailable,
|
|
// Snapshot must still fail rather than silently return an empty match pool.
|
|
func TestCandidateProjectionStillFailsWhenBothRedisAndSourceAreDown(t *testing.T) {
|
|
mini, err := miniredis.Run()
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
client := redis.NewClient(&redis.Options{Addr: mini.Addr()})
|
|
defer client.Close()
|
|
mini.Close()
|
|
|
|
projection := CandidateProjection{
|
|
Index: RedisCandidateIndex{Client: client, Prefix: "down", TTL: time.Minute},
|
|
Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
|
|
return nil, context.DeadlineExceeded
|
|
},
|
|
}
|
|
if _, err := projection.Snapshot(context.Background(), domain.Casual, time.Unix(1000, 0), 1000); err == nil {
|
|
t.Fatal("Snapshot succeeded with both Redis and the durable source unavailable")
|
|
}
|
|
}
|
|
|
|
func TestCandidateProjectionRepairsEmptyIndexFromDurableSource(t *testing.T) {
|
|
mini, err := miniredis.Run()
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
defer mini.Close()
|
|
client := redis.NewClient(&redis.Options{Addr: mini.Addr()})
|
|
defer client.Close()
|
|
now := time.Unix(1000, 0).UTC()
|
|
candidate := domain.Candidate{Playlist: domain.Casual, TicketID: "miss-ticket", PlayerID: "miss-player", EnqueuedAt: now}
|
|
projection := CandidateProjection{
|
|
Index: RedisCandidateIndex{Client: client, Prefix: "miss", TTL: time.Minute},
|
|
Source: func(context.Context, domain.Playlist, time.Time, int) ([]domain.Candidate, error) {
|
|
return []domain.Candidate{candidate}, nil
|
|
},
|
|
}
|
|
got, err := projection.Snapshot(context.Background(), domain.Casual, now, 1000)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if len(got) != 1 || got[0].TicketID != candidate.TicketID {
|
|
t.Fatalf("empty-index repair = %+v", got)
|
|
}
|
|
}
|