fix(multiplayer): stop the matcher worker crashing on a routine no-match pass

Found building the two-player proposal integration test (next commit):
Worker.Run treated ANY RunOnce error as fatal to the whole loop,
including domain.FormFromQueue's "no compatible candidates" -- which
is not a failure, it's the completely routine and expected outcome of
a queue whose currently-waiting players don't share a verified region
yet. Two real players with no common region formed exactly this
shape, and the entire matcher process exited -- taking matchmaking
down for every OTHER player in the same playlist, not just the
incompatible pair, since cmd/matcher runs one process per playlist.
Worse: on a real supervisor restart, the same still-incompatible
candidates are still queued, so it would crash again immediately --
an actual crash loop, not a one-off.

RunOnce's own per-call contract (return an error for source failure,
bad formation, mixed playlist, an incomplete batch, a lost durable
claim) is deliberately tested and unchanged. The fix is entirely in
Run's loop: only the three genuinely static misconfiguration errors
(nil dependencies, unsupported playlist, invalid size -- true on every
future pass just as much as this one, so retrying can never help) now
stop it, via new exported sentinels (ErrWorkerNotConfigured,
ErrUnsupportedPlaylist, ErrInvalidMatcherSize) and errors.Is. Every
other RunOnce error is a single pass's worth of "no match formed this
time" and Run keeps polling.

Covered by two new tests: Run recovering from a first-pass error and
still forming a proposal once the pool becomes viable (a real
concurrent goroutine driving Run, not just calling RunOnce directly --
Run's own loop had no test coverage at all before this), and Run still
stopping immediately on a genuine configuration error. Both clean
across 3 runs with -race.
This commit is contained in:
Josh Creek
2026-09-01 14:26:19 +01:00
parent 01bb9a9431
commit 3168dd9897
2 changed files with 96 additions and 4 deletions
+22 -4
View File
@@ -4,12 +4,27 @@ package matcher
import (
"context"
"errors"
"fmt"
"time"
"github.com/cosmic-clash/cosmic-clash/server/domain"
)
// These three are the only RunOnce failures Run treats as fatal to the whole
// worker: they're static misconfiguration, true on every future pass just as
// much as this one, so retrying cannot help. Every other RunOnce error --
// a source read hiccup, no common region among the current candidate pool,
// a losing race against another matcher replica, an incomplete batch -- is a
// single pass's worth of "no match formed this time," a routine and
// expected steady state that must not take matching down for every other
// player still waiting behind it.
var (
ErrWorkerNotConfigured = errors.New("matcher worker is not configured")
ErrUnsupportedPlaylist = errors.New("unsupported matcher playlist")
ErrInvalidMatcherSize = errors.New("invalid matcher size")
)
type CandidateSource func(context.Context, time.Time, domain.Playlist, int) ([]domain.Candidate, error)
type ProposalCreator interface {
@@ -42,7 +57,10 @@ func (w Worker) Run(ctx context.Context, interval time.Duration) error {
}
for {
if _, err := w.RunOnce(ctx); err != nil {
return err
if errors.Is(err, ErrWorkerNotConfigured) || errors.Is(err, ErrUnsupportedPlaylist) || errors.Is(err, ErrInvalidMatcherSize) {
return err
}
// Not fatal -- fall through and retry next interval.
}
timer := time.NewTimer(interval)
select {
@@ -59,13 +77,13 @@ func (w Worker) Run(ctx context.Context, interval time.Duration) error {
// a stale cache therefore fails safely and can be retried on the next pass.
func (w Worker) RunOnce(ctx context.Context) (bool, error) {
if w.Source == nil || w.Creator == nil || w.Now == nil || w.NextID == nil || w.Prepare == nil {
return false, fmt.Errorf("matcher worker is not configured")
return false, ErrWorkerNotConfigured
}
if w.Playlist != domain.Casual && w.Playlist != domain.Ranked {
return false, fmt.Errorf("unsupported matcher playlist")
return false, ErrUnsupportedPlaylist
}
if w.Size < 2 || w.Size > 6 {
return false, fmt.Errorf("invalid matcher size")
return false, ErrInvalidMatcherSize
}
now := w.Now()
candidates, err := w.Source(ctx, now, w.Playlist, w.Size)
+74
View File
@@ -3,6 +3,7 @@ package matcher
import (
"context"
"errors"
"sync"
"testing"
"time"
@@ -104,3 +105,76 @@ func TestRunOnceRejectsMixedPlaylistAndDuplicateIdentityBatches(t *testing.T) {
t.Fatalf("creator calls=%d", creator.calls)
}
}
// TestRunSurvivesPerPassErrorsAndKeepsRetrying reproduces a real production
// bug found via a live integration test (multiplayer-next.md 8.40): two real
// players queued with no verified common region formed exactly this
// "source succeeds, formation fails" shape, and the matcher process died
// entirely rather than waiting for a compatible batch -- silently taking
// matchmaking down for every other player behind them too, not just the
// incompatible pair. Run must survive a per-pass RunOnce error and try
// again next interval rather than returning immediately.
func TestRunSurvivesPerPassErrorsAndKeepsRetrying(t *testing.T) {
var mu sync.Mutex
attempts := 0
creatorCalls := 0
worker := workerFor(func(context.Context, time.Time, domain.Playlist, int) ([]domain.Candidate, error) {
mu.Lock()
defer mu.Unlock()
attempts++
if attempts == 1 {
// Same failure shape as domain.FormFromQueue's "no compatible
// candidates" -- RunOnce still returns a non-nil error here, only
// Run's handling of it is what this test is about.
return nil, errors.New("no common region")
}
return candidates(), nil
}, ProposalCreatorFunc(func(context.Context, domain.Proposal, map[string]string, time.Time) error {
mu.Lock()
defer mu.Unlock()
creatorCalls++
return nil
}))
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
defer cancel()
done := make(chan error, 1)
go func() { done <- worker.Run(ctx, 5*time.Millisecond) }()
deadline := time.After(1 * time.Second)
for {
mu.Lock()
calls := creatorCalls
mu.Unlock()
if calls > 0 {
break
}
select {
case err := <-done:
t.Fatalf("Run returned early on a per-pass error instead of retrying: %v", err)
case <-deadline:
t.Fatal("Run never recovered from the first pass's error")
default:
time.Sleep(time.Millisecond)
}
}
mu.Lock()
defer mu.Unlock()
if creatorCalls != 1 {
t.Fatalf("creator calls=%d, want exactly 1 once formation finally succeeded", creatorCalls)
}
}
// TestRunStopsImmediatelyOnConfigurationErrors is the other half of the
// fix: a genuinely static misconfiguration (true on every future pass, not
// just this one) must still stop the worker rather than spin forever.
func TestRunStopsImmediatelyOnConfigurationErrors(t *testing.T) {
worker := workerFor(func(context.Context, time.Time, domain.Playlist, int) ([]domain.Candidate, error) {
return candidates(), nil
}, &creatorSpy{})
worker.Playlist = domain.Playlist("invalid")
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Second)
defer cancel()
err := worker.Run(ctx, 5*time.Millisecond)
if !errors.Is(err, ErrUnsupportedPlaylist) {
t.Fatalf("Run() error = %v, want ErrUnsupportedPlaylist", err)
}
}