From 3168dd9897903620d9de303c1abf8b59c600c504 Mon Sep 17 00:00:00 2001 From: Josh Creek <8179928+jcreek@users.noreply.github.com> Date: Tue, 1 Sep 2026 14:26:19 +0100 Subject: [PATCH] fix(multiplayer): stop the matcher worker crashing on a routine no-match pass Found building the two-player proposal integration test (next commit): Worker.Run treated ANY RunOnce error as fatal to the whole loop, including domain.FormFromQueue's "no compatible candidates" -- which is not a failure, it's the completely routine and expected outcome of a queue whose currently-waiting players don't share a verified region yet. Two real players with no common region formed exactly this shape, and the entire matcher process exited -- taking matchmaking down for every OTHER player in the same playlist, not just the incompatible pair, since cmd/matcher runs one process per playlist. Worse: on a real supervisor restart, the same still-incompatible candidates are still queued, so it would crash again immediately -- an actual crash loop, not a one-off. RunOnce's own per-call contract (return an error for source failure, bad formation, mixed playlist, an incomplete batch, a lost durable claim) is deliberately tested and unchanged. The fix is entirely in Run's loop: only the three genuinely static misconfiguration errors (nil dependencies, unsupported playlist, invalid size -- true on every future pass just as much as this one, so retrying can never help) now stop it, via new exported sentinels (ErrWorkerNotConfigured, ErrUnsupportedPlaylist, ErrInvalidMatcherSize) and errors.Is. Every other RunOnce error is a single pass's worth of "no match formed this time" and Run keeps polling. Covered by two new tests: Run recovering from a first-pass error and still forming a proposal once the pool becomes viable (a real concurrent goroutine driving Run, not just calling RunOnce directly -- Run's own loop had no test coverage at all before this), and Run still stopping immediately on a genuine configuration error. Both clean across 3 runs with -race. --- server/matcher/worker.go | 26 ++++++++++-- server/matcher/worker_test.go | 74 +++++++++++++++++++++++++++++++++++ 2 files changed, 96 insertions(+), 4 deletions(-) diff --git a/server/matcher/worker.go b/server/matcher/worker.go index 46259ea1..13d037d1 100644 --- a/server/matcher/worker.go +++ b/server/matcher/worker.go @@ -4,12 +4,27 @@ package matcher import ( "context" + "errors" "fmt" "time" "github.com/cosmic-clash/cosmic-clash/server/domain" ) +// These three are the only RunOnce failures Run treats as fatal to the whole +// worker: they're static misconfiguration, true on every future pass just as +// much as this one, so retrying cannot help. Every other RunOnce error -- +// a source read hiccup, no common region among the current candidate pool, +// a losing race against another matcher replica, an incomplete batch -- is a +// single pass's worth of "no match formed this time," a routine and +// expected steady state that must not take matching down for every other +// player still waiting behind it. +var ( + ErrWorkerNotConfigured = errors.New("matcher worker is not configured") + ErrUnsupportedPlaylist = errors.New("unsupported matcher playlist") + ErrInvalidMatcherSize = errors.New("invalid matcher size") +) + type CandidateSource func(context.Context, time.Time, domain.Playlist, int) ([]domain.Candidate, error) type ProposalCreator interface { @@ -42,7 +57,10 @@ func (w Worker) Run(ctx context.Context, interval time.Duration) error { } for { if _, err := w.RunOnce(ctx); err != nil { - return err + if errors.Is(err, ErrWorkerNotConfigured) || errors.Is(err, ErrUnsupportedPlaylist) || errors.Is(err, ErrInvalidMatcherSize) { + return err + } + // Not fatal -- fall through and retry next interval. } timer := time.NewTimer(interval) select { @@ -59,13 +77,13 @@ func (w Worker) Run(ctx context.Context, interval time.Duration) error { // a stale cache therefore fails safely and can be retried on the next pass. func (w Worker) RunOnce(ctx context.Context) (bool, error) { if w.Source == nil || w.Creator == nil || w.Now == nil || w.NextID == nil || w.Prepare == nil { - return false, fmt.Errorf("matcher worker is not configured") + return false, ErrWorkerNotConfigured } if w.Playlist != domain.Casual && w.Playlist != domain.Ranked { - return false, fmt.Errorf("unsupported matcher playlist") + return false, ErrUnsupportedPlaylist } if w.Size < 2 || w.Size > 6 { - return false, fmt.Errorf("invalid matcher size") + return false, ErrInvalidMatcherSize } now := w.Now() candidates, err := w.Source(ctx, now, w.Playlist, w.Size) diff --git a/server/matcher/worker_test.go b/server/matcher/worker_test.go index ea6a030a..1a5bea46 100644 --- a/server/matcher/worker_test.go +++ b/server/matcher/worker_test.go @@ -3,6 +3,7 @@ package matcher import ( "context" "errors" + "sync" "testing" "time" @@ -104,3 +105,76 @@ func TestRunOnceRejectsMixedPlaylistAndDuplicateIdentityBatches(t *testing.T) { t.Fatalf("creator calls=%d", creator.calls) } } + +// TestRunSurvivesPerPassErrorsAndKeepsRetrying reproduces a real production +// bug found via a live integration test (multiplayer-next.md 8.40): two real +// players queued with no verified common region formed exactly this +// "source succeeds, formation fails" shape, and the matcher process died +// entirely rather than waiting for a compatible batch -- silently taking +// matchmaking down for every other player behind them too, not just the +// incompatible pair. Run must survive a per-pass RunOnce error and try +// again next interval rather than returning immediately. +func TestRunSurvivesPerPassErrorsAndKeepsRetrying(t *testing.T) { + var mu sync.Mutex + attempts := 0 + creatorCalls := 0 + worker := workerFor(func(context.Context, time.Time, domain.Playlist, int) ([]domain.Candidate, error) { + mu.Lock() + defer mu.Unlock() + attempts++ + if attempts == 1 { + // Same failure shape as domain.FormFromQueue's "no compatible + // candidates" -- RunOnce still returns a non-nil error here, only + // Run's handling of it is what this test is about. + return nil, errors.New("no common region") + } + return candidates(), nil + }, ProposalCreatorFunc(func(context.Context, domain.Proposal, map[string]string, time.Time) error { + mu.Lock() + defer mu.Unlock() + creatorCalls++ + return nil + })) + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + defer cancel() + done := make(chan error, 1) + go func() { done <- worker.Run(ctx, 5*time.Millisecond) }() + deadline := time.After(1 * time.Second) + for { + mu.Lock() + calls := creatorCalls + mu.Unlock() + if calls > 0 { + break + } + select { + case err := <-done: + t.Fatalf("Run returned early on a per-pass error instead of retrying: %v", err) + case <-deadline: + t.Fatal("Run never recovered from the first pass's error") + default: + time.Sleep(time.Millisecond) + } + } + mu.Lock() + defer mu.Unlock() + if creatorCalls != 1 { + t.Fatalf("creator calls=%d, want exactly 1 once formation finally succeeded", creatorCalls) + } +} + +// TestRunStopsImmediatelyOnConfigurationErrors is the other half of the +// fix: a genuinely static misconfiguration (true on every future pass, not +// just this one) must still stop the worker rather than spin forever. +func TestRunStopsImmediatelyOnConfigurationErrors(t *testing.T) { + worker := workerFor(func(context.Context, time.Time, domain.Playlist, int) ([]domain.Candidate, error) { + return candidates(), nil + }, &creatorSpy{}) + worker.Playlist = domain.Playlist("invalid") + ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second) + defer cancel() + err := worker.Run(ctx, 5*time.Millisecond) + if !errors.Is(err, ErrUnsupportedPlaylist) { + t.Fatalf("Run() error = %v, want ErrUnsupportedPlaylist", err) + } +}