From 17bd7768ebf679ac9a5fc0e3816a37a7d01ac381 Mon Sep 17 00:00:00 2001 From: Josh Creek <8179928+jcreek@users.noreply.github.com> Date: Tue, 1 Sep 2026 13:20:09 +0100 Subject: [PATCH] test(multiplayer): add real-Redis integration coverage for the candidate index Every existing RedisCandidateIndex test runs against miniredis -- a from-scratch Go reimplementation of the Redis command set, not real Redis's own float64 score encoding, TTL/expiry, or RESP behavior. Add an opt-in integration suite (mirroring postgres_integration_test.go's pattern: //go:build integration, COSMIC_CLASH_REDIS_ADDR-gated) plus scripts/run_redis_integration.sh against a disposable redis:7-alpine container, covering: - upsert/snapshot/remove against a real server - a real TTL actually waited out (not miniredis's manual FastForward), proving expiry really happens on the wire - the documented 'Redis restart or lost keyspace' repair path exercised against an actual FLUSHALL, not a simulated empty map, including that the repair actually persists back to Redis (a second snapshot reads it without a second durable-source call) Verified stable across 3 runs with -race against a real container. --- scripts/run_redis_integration.sh | 29 +++++ server/store/redis_integration_test.go | 147 +++++++++++++++++++++++++ 2 files changed, 176 insertions(+) create mode 100755 scripts/run_redis_integration.sh create mode 100644 server/store/redis_integration_test.go diff --git a/scripts/run_redis_integration.sh b/scripts/run_redis_integration.sh new file mode 100755 index 00000000..950e18e0 --- /dev/null +++ b/scripts/run_redis_integration.sh @@ -0,0 +1,29 @@ +#!/usr/bin/env bash +set -euo pipefail + +repo_root="$(cd "$(dirname "$0")/.." && pwd)" +container_name="cosmic-clash-redis-integration" + +cleanup() { + docker rm -f "$container_name" >/dev/null 2>&1 || true +} +trap cleanup EXIT + +cleanup +docker run --rm -d --name "$container_name" \ + -p 56379:6379 redis:7-alpine >/dev/null + +for attempt in $(seq 1 30); do + if docker exec "$container_name" redis-cli ping >/dev/null 2>&1; then + break + fi + if [ "$attempt" = 30 ]; then + echo "Redis did not become ready" >&2 + exit 1 + fi + sleep 1 +done + +cd "$repo_root/server" +COSMIC_CLASH_REDIS_ADDR="127.0.0.1:56379" \ + go test -tags integration ./store -run TestRealRedis -count=1 diff --git a/server/store/redis_integration_test.go b/server/store/redis_integration_test.go new file mode 100644 index 00000000..fa3f783d --- /dev/null +++ b/server/store/redis_integration_test.go @@ -0,0 +1,147 @@ +//go:build integration + +package store + +import ( + "context" + "os" + "testing" + "time" + + "github.com/cosmic-clash/cosmic-clash/server/domain" + "github.com/redis/go-redis/v9" +) + +// This binary is deliberately opt-in, mirroring postgres_integration_test.go: +// it requires a disposable real Redis supplied by +// scripts/run_redis_integration.sh, as distinct from the miniredis-backed +// unit tests in redis_candidates_test.go and candidate_projection_test.go. +// miniredis is a from-scratch Go reimplementation of the Redis command set -- +// it does not run real Redis's own float64 score encoding, real TTL/expiry, +// or real RESP wire behavior, so it cannot by itself prove this code works +// against the real thing, only that it works against a same-language model of +// it. +func openIntegrationRedis(t *testing.T) *redis.Client { + t.Helper() + addr := os.Getenv("COSMIC_CLASH_REDIS_ADDR") + if addr == "" { + t.Skip("COSMIC_CLASH_REDIS_ADDR is not set") + } + client := redis.NewClient(&redis.Options{Addr: addr}) + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second) + defer cancel() + if err := client.Ping(ctx).Err(); err != nil { + client.Close() + t.Fatalf("ping Redis: %v", err) + } + if err := client.FlushAll(ctx).Err(); err != nil { + client.Close() + t.Fatalf("reset Redis: %v", err) + } + t.Cleanup(func() { client.Close() }) + return client +} + +func TestRealRedisCandidateIndexUpsertSnapshotRemove(t *testing.T) { + client := openIntegrationRedis(t) + ctx := context.Background() + index := RedisCandidateIndex{Client: client, Prefix: "integration-real", TTL: time.Minute} + now := time.Now().UTC().Truncate(time.Microsecond) + + a := domain.Candidate{TicketID: "real-ticket-a", PlayerID: "real-player-a", Playlist: domain.Casual, ClientBuild: "build-1", ProtocolVersion: 1, EnqueuedAt: now} + b := domain.Candidate{TicketID: "real-ticket-b", PlayerID: "real-player-b", Playlist: domain.Casual, ClientBuild: "build-1", ProtocolVersion: 1, EnqueuedAt: now.Add(time.Second)} + if err := index.Upsert(ctx, a); err != nil { + t.Fatalf("upsert a: %v", err) + } + if err := index.Upsert(ctx, b); err != nil { + t.Fatalf("upsert b: %v", err) + } + got, err := index.Snapshot(ctx, now.Add(time.Hour)) + if err != nil { + t.Fatalf("snapshot: %v", err) + } + if len(got) != 2 || got[0].TicketID != "real-ticket-a" || got[1].TicketID != "real-ticket-b" { + t.Fatalf("snapshot after upsert = %+v", got) + } + + if err := index.Remove(ctx, "real-ticket-a"); err != nil { + t.Fatalf("remove: %v", err) + } + got, err = index.Snapshot(ctx, now.Add(time.Hour)) + if err != nil { + t.Fatalf("snapshot after remove: %v", err) + } + if len(got) != 1 || got[0].TicketID != "real-ticket-b" { + t.Fatalf("snapshot after remove = %+v", got) + } + + // A real TTL, actually waited out, not miniredis's manual FastForward. + shortLived := RedisCandidateIndex{Client: client, Prefix: "integration-real-ttl", TTL: 1500 * time.Millisecond} + if err := shortLived.Upsert(ctx, domain.Candidate{TicketID: "real-ticket-ttl", PlayerID: "real-player-ttl", EnqueuedAt: now}); err != nil { + t.Fatalf("upsert ttl candidate: %v", err) + } + time.Sleep(2 * time.Second) + got, err = shortLived.Snapshot(ctx, now.Add(time.Hour)) + if err != nil { + t.Fatalf("snapshot after real TTL expiry: %v", err) + } + if len(got) != 0 { + t.Fatalf("candidate survived its real TTL: %+v", got) + } +} + +// TestRealRedisCandidateProjectionRepairsAfterFlush proves the documented +// "Redis restart or lost keyspace" repair path against an actual data loss +// event on a real server -- FLUSHALL -- not a simulated empty map. +func TestRealRedisCandidateProjectionRepairsAfterFlush(t *testing.T) { + client := openIntegrationRedis(t) + ctx := context.Background() + index := RedisCandidateIndex{Client: client, Prefix: "integration-real-repair", TTL: time.Minute} + now := time.Now().UTC().Truncate(time.Microsecond) + + durable := []domain.Candidate{ + {TicketID: "repair-ticket-a", PlayerID: "repair-player-a", EnqueuedAt: now}, + {TicketID: "repair-ticket-b", PlayerID: "repair-player-b", EnqueuedAt: now.Add(time.Second)}, + } + sourceCalls := 0 + projection := CandidateProjection{Index: index, Source: func(context.Context, time.Time) ([]domain.Candidate, error) { + sourceCalls++ + return durable, nil + }} + + if err := index.Upsert(ctx, durable[0]); err != nil { + t.Fatalf("seed upsert: %v", err) + } + // Simulate the actual failure mode this path exists for: the whole Redis + // instance loses its data (restart without persistence, failover to an + // empty replica, an operator FLUSHALL) mid-operation, not just "this one + // key expired". + if err := client.FlushAll(ctx).Err(); err != nil { + t.Fatalf("flush: %v", err) + } + + got, err := projection.Snapshot(ctx, now.Add(time.Hour)) + if err != nil { + t.Fatalf("snapshot after flush: %v", err) + } + if sourceCalls != 1 { + t.Fatalf("expected exactly one durable repair call, got %d", sourceCalls) + } + if len(got) != 2 || got[0].TicketID != "repair-ticket-a" || got[1].TicketID != "repair-ticket-b" { + t.Fatalf("snapshot after repair = %+v", got) + } + + // The repair must actually have written back to Redis, not just returned + // the durable source's answer in memory -- confirm a second snapshot + // (Redis not flushed again) reads it back without a second Source call. + got, err = index.Snapshot(ctx, now.Add(time.Hour)) + if err != nil { + t.Fatalf("snapshot directly against Redis after repair: %v", err) + } + if len(got) != 2 { + t.Fatalf("repaired data was not actually persisted to Redis: %+v", got) + } + if sourceCalls != 1 { + t.Fatalf("expected repair to persist so a second read needs no further Source call, got %d calls", sourceCalls) + } +}