Files
CosmicClash/Game/scripts/networked_match.gd
T
Josh Creek 2325313ad2 fix(multiplayer): adversarial review fixes for Phase 3
An Opus subagent's adversarial review of Phase 3 found a critical, silent,
permanent bug plus eight smaller real issues, all empirically verified
with real two- and three-process runs:

CRITICAL: InputJitterBuffer's 32-entry ring permanently bricked a
player's input once the un-consumed backlog exceeded the ring's
capacity - a fresh arrival would land in the exact slot consume() was
still waiting on, and since both counters only ever advance, the gap
never closed. Reproduced with a real SIGSTOP/SIGCONT host freeze:
client movement dropped from ~26m to 0.00m at ~0.7s, worse under real
loss (a lossy link lowered the fatal threshold to ~400ms), and
reachable via ordinary clock drift with no external trigger at all.
Fixed by tracking the highest seq ever ingested and having consume()
jump directly to what the ring can still provide once the gap exceeds
capacity, instead of starving through an unrecoverable span. Re-verified
with a 3s freeze (well past the original threshold): full recovery.

HIGH: InputLeadController's release logic was gated on its own past
attacks (lead > LEAD_MIN) rather than the real server-reported depth, so
a backlog it didn't itself cause was never drained. Fixed to gate on
actual depth vs target.

MEDIUM-HIGH: the rate limiter's "N consecutive over-budget seconds"
streak hard-reset to 0 on any clean window, letting a duty-cycled flood
(burst, one clean window, repeat) sustain ~33x budget indefinitely with
zero warnings. Replaced with a leaky-bucket accumulator immune to the
same evasion by construction.

MEDIUM: the seq > server_tick + 20 guard compared two unrelated clock
epochs (server process uptime vs. client's own from-zero seq numbering),
so it never actually protected anything on a long-running server and
could silently drop an honest client's input forever. Bound against the
buffer's own last_applied_seq instead.

MEDIUM: InputJitterBuffer.stalled was computed but never reached the
wire - the one signal that would have made the ring-overflow bug visible
anywhere. Now wired through _ship_to_net_body_state.

MEDIUM: task 3.6's CI driver's assertions didn't depend on client input
reaching the server at all, so it kept passing with the ring-overflow
bug actively triggered. Added real ship-movement and non-stalled checks,
sampled while bots are still connected (an initial attempt sampled after
their own legitimate disconnect, which starves identically to the bug).

LOW-MEDIUM: a lead change silently mislabelled _input_history's older
entries, since the wire format has no per-entry seq field. Fixed by
handling each delta case (ordinary/release/attack) on its own terms.

LOW: bandwidth and snapshot-loss overlay metrics froze at their last
value during a total outage instead of decaying - exactly when they
matter most. Both now report honest post-outage values.

LOW: a guard comment on NetworkManager._ping misdescribed the actual
disconnect_peer() arguments in use. Corrected.

New permanent regression tests: test_ring_overflow_resyncs_to_fresh_data
_instead_of_starving_forever, test_release_drains_a_backlog_it_never_
caused_itself, and client-abuse-flood-dutycycle (reproduces the exact
duty-cycle evasion). Full regression suite, including the net-sim-latency
milestone gate, all abuse roles, and the CI driver, re-run clean after
every fix.
2026-08-20 15:28:44 +01:00

695 lines
34 KiB
GDScript

class_name NetworkedMatch
extends GameMode
# Phase 2: server-authoritative simulation, dumb client (multiplayer-todo.md
# §7 Phase 2). The server runs the real physics for every ship — via
# RLShipController, fed by each connected player's forwarded input — and
# the ball, and broadcasts NetCodec snapshots at 60Hz. The client renders
# everything, including its own ship, from the interpolation buffer; there
# is no local prediction yet (that's Phase 4), so every body on the client
# is FREEZE_MODE_KINEMATIC and driven entirely by incoming snapshots.
#
# No HUD/Arena child in networked_match.tscn — both are built in code, once
# the arena is actually known (the server picks one; the client learns it
# from match_config), which is why this overrides _ready() completely
# rather than relying on GameMode's default (arena-required-synchronously)
# flow.
# Only score_changed is actually emitted in Phase 2 — Phase 5 owns the match
# lifecycle state machine (timer, kickoff countdown, overtime, results), so
# those signals get declared there, alongside real emission. Declaring one
# here without emitting it isn't harmless: HUDController gates the timer
# widget's visibility purely on has_signal("timer_updated"), so a declared-
# but-dead signal shows a permanently frozen timer rather than correctly
# hiding it the way free_play.gd's total absence of the signal does.
signal score_changed(score: Dictionary)
const NetCodec = preload("res://scripts/net_codec.gd")
const NetBodyState = preload("res://scripts/net_body_state.gd")
const NetInterpolator = preload("res://scripts/net_interpolator.gd")
const InputJitterBuffer = preload("res://scripts/input_jitter_buffer.gd")
const InputLeadController = preload("res://scripts/input_lead_controller.gd")
const HUD_SCENE = preload("res://scenes/HUD.tscn")
# Minimum plausible interpolation delay even on a same-machine/LAN link —
# §4.6's INTERP_DELAY clamp floor. The full formula (one_way + snapshot
# interval*1.5 + 2.5*jitter_ewma) is simplified here to one_way + interval*1.5
# with no jitter term yet (no jitter EWMA is tracked before Phase 3) — close
# enough for Phase 2's "smooth, not exactly latency-optimal" bar.
const INTERP_DELAY_MIN_MS := 25.0
const INTERP_DELAY_MAX_MS := 200.0
const SNAPSHOT_INTERVAL_MS := 1000.0 / 60.0
# NetInterpolator.to_tick() assumes Time.get_ticks_msec() == physics_frame *
# TICK_MS on the SERVER, i.e. that physics frame 0 happened at process-start
# wall time. It doesn't: real startup work (autoloads, asset loading) elapses
# before the first physics step, and any dropped tick widens the gap further
# — it only ever grows. An adversarial review found this was NOT a rounding
# error: it measured a steady +45-50ms bias on a real run, meaning EVERY
# to_tick(get_server_time_estimate_ms()) call landed 3+ ticks past the
# newest buffered sample, so sample_at() took the extrapolation branch 100%
# of the time — zero real interpolation ever happened, on LAN or under
# simulated latency alike, silently defeating the entire interpolation
# buffer this phase was built around.
#
# Fix: this bias is a property of the server's clock, not of any one body,
# so track ONE shared estimate here (not per-interpolator) from every
# snapshot's own server_tick versus this client's server-time estimate at
# receipt. Take the MINIMUM over a rolling window — same rationale as
# NetworkManager's own min-RTT filtering (network_manager.gd): the sample
# with the least one-way transit delay best isolates the constant epoch
# bias from per-packet network noise, and a rolling (not all-time) window
# lets a real increase in the bias — the server dropping more ticks later
# in the match — still get picked up rather than staying pinned to a
# now-stale historical minimum.
const TICK_BIAS_WINDOW_SEC := 5.0
var _tick_bias_samples: Array[Dictionary] = [] # [{t_ms:int, bias_ms:float}], client only
var _tick_bias_ms := 0.0 # best current estimate; 0.0 until the first snapshot
class SlotInfo:
var peer_id: int
var team: int
var spawn_index: int
var ship: Ship
var controller: RLShipController # server only
var jitter_buffer := InputJitterBuffer.new() # server only (§3.2)
var last_client_send_ms := 0 # server only: echoed back per-peer next snapshot (§2.4)
var interpolator := NetInterpolator.new() # client only
var _slots: Array[SlotInfo] = []
var _my_slot: SlotInfo = null # client only
var _ball_interpolator := NetInterpolator.new() # client only
# client only: reads local input each tick to forward. Normally a
# PlayerShipController that's deliberately never added to a Ship/the tree —
# get_action() only touches the global Input singleton, so it needs no
# scene context. --test-bot mode (task 3.6) swaps this for a real
# AIShipController once the client's own ship is known (see
# _on_match_config_received) — unlike PlayerShipController, AIShipController
# DOES need real scene context (get_parent() as Ship, plus ball/teammate/
# opponent discovery via groups), so it's parented onto _my_slot.ship via
# Ship.set_controller() rather than left floating.
var _local_input_sampler: ShipController = PlayerShipController.new()
# --test-bot (task 3.6): CI/regression driver mode, an automated player via
# the existing AIShipController instead of a human — see CLAUDE.md's testing
# section. Read once in _ready(), consumed in _on_match_config_received.
var _test_bot_model_path := "" # client only; non-empty means --test-bot mode is active
var _input_seq := 0 # client only
# Redundancy (§3.1): newest-first, capped at NetCodec.MAX_REDUNDANCY, so a
# 3-packet burst loss still recovers every tick's action via a later
# packet's history. Client only.
var _input_history: Array[ShipAction] = []
var _last_received_snapshot_tick := 0 # client only: echoed back as ack_snapshot_tick
var _input_lead_controller := InputLeadController.new() # client only (§3.3)
var _last_known_input_buffer_depth := -1 # client only: -1 = no snapshot with this field yet
# Loss estimate (task 3.7's debug overlay), client only: snapshots go out
# at a steady one-tick cadence, so a server_tick that jumps by more than 1
# since the last received one is direct evidence of a dropped or reordered
# snapshot on the unreliable channel. EWMA over each reception's own
# "missed / (missed + 1)" fraction rather than a flat drop-count, so it
# reads as a live percentage and decays naturally once loss stops.
const SNAPSHOT_LOSS_EWMA_ALPHA := 1.0 / 16.0
var _snapshot_loss_ewma := 0.0
var _expected_next_snapshot_tick := -1
# An adversarial review found _snapshot_loss_ewma only updates on receipt —
# during a TOTAL outage, exactly when this metric matters most, it freezes
# at its last (probably low/healthy) value instead of climbing toward
# 100%. Track wall-clock receipt time so get_net_debug_stats() can report
# honestly once too long has passed with nothing arriving at all.
var _last_snapshot_wall_ms := -1
const SNAPSHOT_STALE_MS := 500.0 # ~30 ticks with nothing at all — treat as total loss, not "still fine"
var _unknown_sender_input_count := 0 # server only, observability (§3.1 step 1)
var _reset_gen := 0 # server only: bumped on every kickoff/goal reset so the client hard-snaps instead of interpolating across the teleport
# Server only. _on_goal_scored's reset_ball()/reset_ships() only QUEUE
# teleports (task 0.15's queue_teleport — applied on each body's next
# _integrate_forces), but _broadcast_snapshot runs later in the SAME frame
# _on_goal_scored fires in, before that teleport lands. Bumping _reset_gen
# immediately would tag the still-pre-teleport snapshot with the new
# generation: the client clears its buffer expecting a hard snap, then
# keeps exactly that stale in-goal sample and lerps a full-arena slide to
# the next, genuinely-post-teleport sample — an adversarial review measured
# a 26.8m ball slide from this.
#
# A plain "bump on the next _physics_process" boolean flag turned out NOT
# to fix it: the goal Area's body_entered signal (and so _on_goal_scored)
# fires as part of physics tick N's OWN step processing, before tick N's
# _physics_process callback — so a flag set there is already true by the
# time that SAME tick's _physics_process checks it, consuming on tick N
# instead of N+1 as intended (empirically confirmed: with a boolean flag,
# gen still bumped on the same tick the stale position was broadcast).
# The queued teleport, by contrast, isn't applied until tick N+1's
# _integrate_forces. So the two must be compared by TICK NUMBER, not by
# "next callback": only bump once the current tick is strictly later than
# the tick the goal was detected on, which guarantees at least one full
# _integrate_forces has run — and therefore the queued teleport has
# landed — since the flag was set.
var _pending_reset_gen_bump := false
var _pending_reset_gen_bump_tick := -1
func _ready() -> void:
add_to_group("game")
Engine.max_physics_steps_per_frame = 4
if kickoff_rng_seed == 0:
_kickoff_rng.randomize()
if multiplayer.is_server():
_start_server()
else:
for arg: String in OS.get_cmdline_user_args():
if arg == "--test-bot":
_test_bot_model_path = "res://bots/promoted/medium.json"
elif arg.begins_with("--test-bot-model="):
_test_bot_model_path = arg.get_slice("=", 1)
MatchSim.match_config_received.connect(_on_match_config_received)
MatchSim.snapshot_received.connect(_on_snapshot_received)
MatchSim.score_update_received.connect(_on_score_update_received)
_request_match_config_until_received()
# The one-shot server broadcast in _start_server() is racy against however
# long this client's own scene load took to reach this line — it may have
# already fired into a MatchSim with no listener connected yet, or the
# server may not have even started the match yet. Keep asking until
# _on_match_config_received actually populates _slots.
func _request_match_config_until_received() -> void:
while _slots.is_empty() and is_inside_tree():
MatchSim.request_match_config()
await get_tree().create_timer(0.5).timeout
func _owns_goal_logic() -> bool:
return multiplayer.is_server()
func _owns_world_simulation() -> bool:
return multiplayer.is_server()
# _local_input_sampler is a plain Node (PlayerShipController extends
# ShipController extends Node) that's deliberately never added to the tree
# — dropping the last reference to it does not free it. An adversarial
# review traced the "3 resources still in use at exit" warning on every
# Phase 2 test run directly to this: --verbose named the leaked script
# chain (player_ship_controller.gd, ship_controller.gd, ship_action.gd)
# exactly, and adding this cleanup made the warning disappear. Runs
# unconditionally (not just client-side) since the field is initialized
# unconditionally too, despite its "client only" comment.
func _exit_tree() -> void:
if is_instance_valid(_local_input_sampler):
_local_input_sampler.free()
# ============================================================
# Server
# ============================================================
func _start_server() -> void:
var arena_path := ArenaRegistry.random_path()
arena = (load(arena_path) as PackedScene).instantiate()
add_child(arena)
for goal in arena.get_goals():
goal.goal_scored.connect(_handle_goal_scored)
spawn_ball()
var peer_ids := PackedInt32Array()
var teams := PackedInt32Array()
var spawn_indices := PackedInt32Array()
var team_counts := {0: 0, 1: 0}
var sorted_peer_ids: Array = MatchNet.roster.keys()
sorted_peer_ids.sort()
for peer_id in sorted_peer_ids:
var info: MatchNet.PlayerInfo = MatchNet.roster[peer_id]
var spawn_index: int = team_counts.get(info.team, 0)
team_counts[info.team] = spawn_index + 1
var slot := SlotInfo.new()
slot.peer_id = peer_id
slot.team = info.team
slot.spawn_index = spawn_index
slot.controller = RLShipController.new()
slot.ship = spawn_ship(info.team, spawn_index, slot.controller)
_slots.append(slot)
peer_ids.append(peer_id)
teams.append(info.team)
spawn_indices.append(spawn_index)
MatchSim.send_match_config(arena_path, peer_ids, teams, spawn_indices)
MatchSim.input_received.connect(_on_input_received)
func _on_input_received(peer_id: int, decoded: Dictionary) -> void:
for slot in _slots:
if slot.peer_id == peer_id:
var seq: int = decoded["seq"]
# §3.1 step 4, rebound after an adversarial review found the
# original check (seq > Engine.get_physics_frames() + 20)
# compared two unrelated epochs: get_physics_frames() counts
# from the SERVER PROCESS's own start, while a client's
# _input_seq starts at 0 when ITS match scene loads —
# input_jitter_buffer.gd's own seeding logic exists specifically
# because these share no baseline (see its header comment).
# Bounding against server uptime meant this guard could never
# fire on a long-running dedicated server (no real protection —
# the stated "keeps garbage-far-future seq values out of the
# ring" rationale wasn't actually achieved), and could silently
# drop an honest client's input forever the moment accumulated
# server tick loss closed whatever accidental head-start margin
# existed. Bound against this slot's own last_applied_seq
# instead — the client's own epoch, which the ring is already
# anchored to — using the ring's own capacity as the bound,
# exactly matching what InputJitterBuffer.consume()'s own
# overflow-resync logic treats as "unrecoverably far ahead"
# anyway. Falls back to seq itself (never rejects) before the
# buffer has ever been seeded — there's no baseline yet to
# bound against.
var jb := slot.jitter_buffer
var seq_bound: int = (jb.last_applied_seq if jb.last_applied_seq >= 0 else seq) + InputJitterBuffer.RING_SIZE
if seq > seq_bound:
return
jb.ingest(seq, decoded["actions"])
slot.last_client_send_ms = decoded["client_send_ms"]
return
# A connected-but-not-yet-slotted peer (or one whose slot somehow
# vanished) sending input — harmless (the packet is simply dropped,
# same as always), but worth counting for observability (§3.1 step 1)
# rather than silently discarding with no trace at all.
_unknown_sender_input_count += 1
func _on_goal_registered(conceding_team: int) -> void:
_record_goal(1 - conceding_team)
MatchSim.send_score_update(score.duplicate())
func _on_goal_scored(_conceding_team: int) -> void:
reset_ball()
reset_ships()
_pending_reset_gen_bump = true
_pending_reset_gen_bump_tick = Engine.get_physics_frames()
func _broadcast_snapshot() -> void:
var server_tick := Engine.get_physics_frames()
var bodies: Array[NetBodyState] = []
# Always one entry per slot, even for a momentarily-invalid ship
# (placeholder zero state), so the ball always lands at the fixed index
# _slots.size() the client assumes in _on_snapshot_received — skipping
# invalid ships entirely would shift every later index. "No ship is ever
# despawned" (§6.4) means this is unreachable today, but it's a silent
# total-garbage failure mode the moment that stops being true, and the
# fix costs nothing.
for slot in _slots:
bodies.append(_ship_to_net_body_state(slot.ship, slot.jitter_buffer.stalled) if is_instance_valid(slot.ship) else NetBodyState.new())
if is_instance_valid(ball):
bodies.append(_ball_to_net_body_state(ball))
var segment := NetCodec.pack_snapshot_body_segment(server_tick, 0, _reset_gen, bodies)
# Building the shared body segment once and reusing it per peer (rather
# than re-encoding per client) is the whole reason §2.4 splits the wire
# format into a per-client header + a shared body segment in the first
# place — see pack_snapshot_body_segment's own doc comment. The per-
# client header (last_input_seq/input_buffer_depth/echo_client_send_ms)
# is genuinely per-peer, built fresh below from each slot's own
# InputJitterBuffer (§3.2) — last_applied_seq of -1 (nothing consumed
# yet) encodes as 0 on the wire, which is safe: the client's own seq
# numbering starts at 1, so 0 never collides with a real seq.
# "No ship is ever despawned" (§6.4) means _slots outlives a disconnect —
# a real one will be handled by Phase 5's reconnect/controller-swap
# logic, but sending an RPC to a peer_id ENet no longer knows about
# (found via the smoke test: a client that exits mid-match spammed
# "Attempt to call RPC with unknown peer ID" every tick for the rest of
# the host's run) throws instead of silently no-op'ing. Guard against it.
var connected_peers := multiplayer.get_peers()
for slot in _slots:
if connected_peers.has(slot.peer_id):
var last_input_seq := maxi(slot.jitter_buffer.last_applied_seq, 0)
var bytes := NetCodec.pack_snapshot(last_input_seq, slot.jitter_buffer.depth(), slot.last_client_send_ms, segment)
MatchSim.send_snapshot(slot.peer_id, bytes)
func _ship_to_net_body_state(ship: Ship, stalled: bool) -> NetBodyState:
var s := NetBodyState.new()
s.position = ship.global_position
s.rotation = ship.global_transform.basis.get_rotation_quaternion()
s.linear_velocity = ship.linear_velocity
s.angular_velocity = ship.angular_velocity
s.frozen = false
s.turbo = ship.is_turbo_active()
# Matches Ship._update_movement_vfx's own read of thrust.z: only positive
# forward thrust drives the visible flame (see task 2.6).
s.thrust_z = clampf(maxf(ship.controller.get_action().thrust.z if ship.controller else 0.0, 0.0), 0.0, 1.0)
s.avel_range = NetCodec.SHIP_AVEL_RANGE
# §3.2: InputJitterBuffer.stalled was computed all along but never
# reached the wire — an adversarial review found this was the exact
# signal that would have made the ring-overflow bug (this session's
# critical fix) visible to the client, the debug overlay, and the CI
# gate, and its absence is part of why none of them ever noticed.
s.stalled = stalled
return s
func _ball_to_net_body_state(b: RigidBody3D) -> NetBodyState:
var s := NetBodyState.new()
s.position = b.global_position
s.rotation = b.global_transform.basis.get_rotation_quaternion()
s.linear_velocity = b.linear_velocity
s.angular_velocity = b.angular_velocity
s.avel_range = NetCodec.BALL_AVEL_RANGE
return s
# ============================================================
# Client
# ============================================================
func _on_match_config_received(arena_path: String, peer_ids: PackedInt32Array, teams: PackedInt32Array, spawn_indices: PackedInt32Array) -> void:
if not _slots.is_empty():
# Not idempotent by accident: the original broadcast from
# _start_server() and a reply to this client's own
# request_match_config() (see _request_match_config_until_received)
# can both legitimately arrive — the retry loop exists specifically
# because either one alone isn't reliably delivered, so seeing both
# is expected, not a protocol error. Processing this twice would
# double-spawn the whole match (found via the two-process smoke
# test: two arenas, two ships, two HUDs, _slots.size() == 2 instead
# of 1). Once is enough.
return
var known := false
for a in ArenaRegistry.ARENAS:
if a["path"] == arena_path:
known = true
break
if not known:
push_error("NetworkedMatch: server sent unknown arena path '%s', refusing match_config" % arena_path)
return
arena = (load(arena_path) as PackedScene).instantiate()
add_child(arena)
# _owns_goal_logic() is false here, so GameMode's usual goal-signal wiring
# never happens — a client's local (interpolated, laggy) Goal sensor must
# never be allowed to decide a score, only the server's real one can.
spawn_ball()
ball.freeze = true
ball.freeze_mode = RigidBody3D.FREEZE_MODE_KINEMATIC
var my_id := multiplayer.get_unique_id()
for i in peer_ids.size():
var slot := SlotInfo.new()
slot.peer_id = peer_ids[i]
slot.team = teams[i]
slot.spawn_index = spawn_indices[i]
slot.ship = spawn_ship(slot.team, slot.spawn_index, null)
slot.ship.freeze = true
slot.ship.freeze_mode = RigidBody3D.FREEZE_MODE_KINEMATIC
# §4.6: manual, per-render-frame $Visual updates must not fight
# Godot's own built-in physics interpolation.
if is_instance_valid(slot.ship.visual):
slot.ship.visual.physics_interpolation_mode = Node.PHYSICS_INTERPOLATION_MODE_OFF
_slots.append(slot)
if slot.peer_id == my_id:
_my_slot = slot
_spawn_hud()
if is_instance_valid(_my_slot) and is_instance_valid(_my_slot.ship):
spawn_camera_rig(_my_slot.ship)
if not _test_bot_model_path.is_empty():
# --test-bot (task 3.6): swap the human input sampler for a real
# AIShipController. Unlike PlayerShipController, this one needs
# real scene context (get_parent() as Ship for itself, plus
# ball/teammate/opponent discovery via groups) — Ship.set_controller()
# parents it correctly, satisfying that. Known limitation: this
# client's ships are all FREEZE_MODE_KINEMATIC and driven purely by
# transform writes (§4.1/§4.6) — nothing here ever writes
# linear_velocity/angular_velocity onto them, so ShipObservations
# always sees every ship (including this one's own) as
# stationary. The policy still produces well-formed, bounded
# actions from that degraded input (PolicyNetwork's output layer
# is bounded regardless of input quality) — good enough for a CI
# traffic generator, which is this task's actual job, not bot
# skill.
var bot := AIShipController.new()
bot.model_path = _test_bot_model_path
_my_slot.ship.set_controller(bot)
# Reassigning _local_input_sampler would orphan the original
# PlayerShipController it pointed to — the exact same leak class
# an adversarial review already caught once for this same field
# (it's a plain Node, never in the tree, so nothing else would
# ever free it). It's never parented, so free() is safe directly.
if is_instance_valid(_local_input_sampler):
_local_input_sampler.free()
_local_input_sampler = bot
func _spawn_hud() -> void:
hud = HUD_SCENE.instantiate()
add_child(hud)
func _send_local_input() -> void:
if _slots.is_empty():
return # match_config hasn't arrived yet
var action := _local_input_sampler.get_action().copy()
# Client-owned input_lead control loop (§3.3): ordinarily +1 (ship
# increments its send sequence by exactly one tick's worth), but a lead
# change this tick skips extra sequence numbers (attack, more server-
# side buffer margin) or duplicates the current one (release, delta 0 —
# one tick of latency recovered).
var delta := _input_lead_controller.update(_last_known_input_buffer_depth)
_input_seq += delta
# Redundancy (§3.1): carry the last MAX_REDUNDANCY ticks' actions,
# newest-first, so a burst of up to (MAX_REDUNDANCY - 1) consecutive
# packet losses still lets the server recover every dropped tick's
# action from a later packet — InputJitterBuffer.ingest() discards
# whichever of these the server already applied, so re-sending old
# ticks every packet is harmless, not just tolerated. NetCodec's wire
# format has no per-entry seq field — actions[i] is implicitly
# "seq - i" — so _input_history must actually BE that many consecutive
# ticks, not just "the last few samples taken". A plain push_front on
# every tick regardless of delta broke that: an adversarial review
# found a lead change silently relabelled older entries (a duplicated
# tick shifts everything back by one position without a matching seq
# change, and a skip-ahead makes the whole history discontiguous with
# the new seq), causing the server to replay already-applied ticks or
# apply the wrong redundant copy for a given seq. Handle each case on
# its own terms instead of always pushing.
if delta == 1:
_input_history.push_front(action)
if _input_history.size() > NetCodec.MAX_REDUNDANCY:
_input_history.resize(NetCodec.MAX_REDUNDANCY)
elif delta == 0:
# Release: seq didn't advance, so this tick's freshest sample
# REPLACES the front entry (still "seq") rather than pushing
# everything else back a position under a label that no longer
# matches what's actually there.
if _input_history.is_empty():
_input_history.push_front(action)
else:
_input_history[0] = action
else:
# Attack: seq jumped ahead by more than one, so nothing previously
# in history is contiguous with the new seq any more — the skipped
# range was never sent, by design (that's what "buys more server-
# side buffer margin" means). Reset the redundancy window to just
# this tick's sample; it rebuilds naturally over the next few
# ticks, the same way it does at connection start.
_input_history = [action]
var bytes := NetCodec.pack_input(_input_seq, _last_received_snapshot_tick, Time.get_ticks_msec(), _input_history)
MatchSim.send_input(bytes)
func _on_snapshot_received(decoded: Dictionary) -> void:
var server_tick: int = decoded["server_tick"]
var reset_gen: int = decoded["reset_gen"]
var bodies: Array = decoded["bodies"]
if _expected_next_snapshot_tick >= 0:
var missed := maxi(0, server_tick - _expected_next_snapshot_tick)
var sample := float(missed) / float(missed + 1)
_snapshot_loss_ewma += (sample - _snapshot_loss_ewma) * SNAPSHOT_LOSS_EWMA_ALPHA
_expected_next_snapshot_tick = server_tick + 1
_last_received_snapshot_tick = server_tick
_last_snapshot_wall_ms = Time.get_ticks_msec()
# Per-client header (§2.4): unlike the shared body segment, this is
# genuinely this recipient's own — input_buffer_depth is THIS client's
# own slot's server-side InputJitterBuffer.depth() at send time, which
# is exactly what the input_lead control loop (§3.3) needs.
_last_known_input_buffer_depth = decoded["input_buffer_depth"]
_update_tick_bias(server_tick)
for i in _slots.size():
if i < bodies.size():
_slots[i].interpolator.add_sample(server_tick, bodies[i], reset_gen)
if bodies.size() > _slots.size():
var ball_state: NetBodyState = bodies[_slots.size()]
# unpack_snapshot() decodes every body's angular_velocity assuming
# SHIP_AVEL_RANGE; the ball was quantised at BALL_AVEL_RANGE
# (_ball_to_net_body_state), so it decodes 8x too small without this
# — dormant today (nothing reads decoded angular_velocity yet) but
# silently wrong the moment ball-spin VFX or Phase 4 prediction does.
NetCodec.rescale_avel(ball_state, NetCodec.BALL_AVEL_RANGE)
_ball_interpolator.add_sample(server_tick, ball_state, reset_gen)
# See the class-level comment above _tick_bias_samples for why this exists.
# bias_ms is how much further ahead to_tick(server_time_est) lands than the
# server_tick this snapshot actually carries — mostly the server's own
# physics-frame/wall-clock startup skew, plus a little real one-way transit
# noise that the rolling minimum below filters back out.
func _update_tick_bias(server_tick: int) -> void:
# get_server_time_estimate_ms() is meaningless before the first pong
# lands (clock_offset_ms == 0.0 until then, per network_manager.gd's own
# doc comment) — recording a bias sample from it during that window
# produced a garbage value (~-1.1s, the client's own raw pre-sync
# uptime standing in for a server-synced estimate) that the rolling-min
# window then locked onto for the rest of a short test, since 5 real
# seconds never fully elapsed before the test ended. Skip entirely
# until the clock is actually synced.
if NetworkManager.rtt_ms < 0.0:
return
var server_time_est := NetworkManager.get_server_time_estimate_ms()
var bias_ms := server_time_est - float(server_tick) * NetInterpolator.TICK_MS
var now_ms := Time.get_ticks_msec()
_tick_bias_samples.append({"t_ms": now_ms, "bias_ms": bias_ms})
var cutoff := now_ms - int(TICK_BIAS_WINDOW_SEC * 1000.0)
_tick_bias_samples = _tick_bias_samples.filter(func(s: Dictionary) -> bool: return s["t_ms"] >= cutoff)
var best: float = _tick_bias_samples[0]["bias_ms"]
for sample: Dictionary in _tick_bias_samples:
var sample_bias: float = sample["bias_ms"]
if sample_bias < best:
best = sample_bias
_tick_bias_ms = best
# Bias-corrected replacement for NetInterpolator.to_tick(server_time_est) —
# use this instead of calling to_tick() directly on a server-time estimate.
func _estimated_tick(server_time_ms: float) -> float:
return NetInterpolator.to_tick(server_time_ms - _tick_bias_ms)
func _current_interp_delay_ms() -> float:
var rtt := NetworkManager.rtt_ms
var one_way := (rtt / 2.0) if rtt >= 0.0 else INTERP_DELAY_MIN_MS
return clampf(one_way + SNAPSHOT_INTERVAL_MS * 1.5, INTERP_DELAY_MIN_MS, INTERP_DELAY_MAX_MS)
# Client-only stats for task 3.7's debug overlay, discovered via the "game"
# group the same way HUDController finds this node — no direct reference
# needed, and the overlay degrades gracefully (has_method check) against
# any mode that doesn't implement this at all.
func get_net_debug_stats() -> Dictionary:
var snapshot_age_ms := 0.0
if NetworkManager.rtt_ms >= 0.0:
var estimated_now_tick := _estimated_tick(NetworkManager.get_server_time_estimate_ms())
snapshot_age_ms = (estimated_now_tick - float(_last_received_snapshot_tick)) * NetInterpolator.TICK_MS
# _snapshot_loss_ewma only updates on receipt, so during a TOTAL outage
# — exactly when this matters most — it would otherwise freeze at
# whatever it last read (probably low/healthy) instead of climbing
# toward 100%, an adversarial review found. Report honestly once too
# long has passed with nothing arriving at all.
var is_stale := _last_snapshot_wall_ms >= 0 and Time.get_ticks_msec() - _last_snapshot_wall_ms > SNAPSHOT_STALE_MS
var snapshot_loss_pct := 100.0 if is_stale else _snapshot_loss_ewma * 100.0
return {
"input_buffer_depth": _last_known_input_buffer_depth,
"input_lead": _input_lead_controller.lead,
"snapshot_age_ms": snapshot_age_ms,
"snapshot_loss_pct": snapshot_loss_pct,
}
# Collider time: present-time estimate, applied once per physics tick.
func _physics_process(_delta: float) -> void:
# Automatic multiplayer polling is disabled project-wide (task 1.3) —
# every scene that sends/receives RPCs has to poll manually, and this
# one is no exception. Missing this meant NOTHING sent after entering
# this scene ever actually reached the wire in either direction
# (queued but never flushed) — found via the two-process smoke test,
# not by inspection.
NetworkManager.poll()
if _owns_world_simulation():
_respawn_escaped_bodies()
if multiplayer.is_server():
# Once per tick, before the step (§3.2) — RLShipController reads
# .action lazily in the ship's own _integrate_forces, which for this
# tick already ran (physics step precedes _physics_process, §9
# gotcha 34), so this actually takes effect on the NEXT tick's step.
# That's the same one-tick input latency Phase 2 already had; this
# just replaces "read the newest packet naively" with a real
# sequence-tracked ring buffer that survives redundant/reordered/
# lost packets.
for slot in _slots:
slot.controller.action = slot.jitter_buffer.consume()
if _pending_reset_gen_bump and Engine.get_physics_frames() > _pending_reset_gen_bump_tick:
_reset_gen = (_reset_gen + 1) % 256
_pending_reset_gen_bump = false
_broadcast_snapshot()
return
_send_local_input()
# get_server_time_estimate_ms() is meaningless before the first pong
# lands (network_manager.gd's own doc comment says so explicitly) — an
# adversarial review found this was used unguarded here, which against
# a long-running dedicated server (clock_offset_ms == 0.0, so this
# process's own short uptime is compared against the server's enormous
# tick count) freezes every remote body at the oldest buffered pose for
# the whole first second of every match.
if NetworkManager.rtt_ms < 0.0:
return
var server_time_est := NetworkManager.get_server_time_estimate_ms()
var collider_tick := _estimated_tick(server_time_est)
for slot in _slots:
if is_instance_valid(slot.ship) and slot.interpolator.has_samples():
_apply_collider_state(slot.ship, slot.interpolator.sample_at(collider_tick))
if is_instance_valid(ball) and _ball_interpolator.has_samples():
_apply_collider_state(ball, _ball_interpolator.sample_at(collider_tick))
# Visual time: present-minus-INTERP_DELAY, applied once per rendered frame —
# separate from the collider update above so a high-refresh client samples
# remote motion at true render rate instead of repeating the same 60Hz value
# several times in a row (§2.4's "240 distinct positions/s, not 60").
#
# Ball only gets the VFX half of this (trail speed), not a transform write:
# unlike Ship, Ball has no separate $Visual child to offset from its
# collider (task 0.2's Visual-node split was scoped to Ship only) — giving
# it one is a bigger structural change than Phase 2's remit, so for now the
# ball's rendered position is whatever _physics_process's present-time
# collider update leaves it at, one tick behind true dual-time smoothness.
func _process(_delta: float) -> void:
# §7 task 1.3: poll for receive unconditionally at the top of both
# _process and _physics_process, not just physics — a snapshot that
# lands between ticks can be rendered immediately at high refresh rates
# instead of waiting for the next physics step.
NetworkManager.poll()
if multiplayer.is_server() or _slots.is_empty():
return
if NetworkManager.rtt_ms < 0.0:
return
var server_time_est := NetworkManager.get_server_time_estimate_ms()
var visual_tick := _estimated_tick(server_time_est - _current_interp_delay_ms())
for slot in _slots:
if is_instance_valid(slot.ship) and slot.interpolator.has_samples():
_apply_ship_visual_state(slot.ship, slot.interpolator.sample_at(visual_tick))
if is_instance_valid(ball) and _ball_interpolator.has_samples():
var state := _ball_interpolator.sample_at(visual_tick)
if state != null:
(ball as Ball).set_visual_speed(state.linear_velocity.length())
func _apply_collider_state(body: RigidBody3D, state: NetBodyState) -> void:
if state == null:
return
body.global_transform = Transform3D(Basis(state.rotation), state.position)
func _apply_ship_visual_state(ship: Ship, state: NetBodyState) -> void:
if state == null:
return
if is_instance_valid(ship.visual):
ship.visual.global_transform = Transform3D(Basis(state.rotation), state.position)
ship.set_visual_action(state.thrust_z, state.turbo)
func _on_score_update_received(new_score: Dictionary) -> void:
score = new_score.duplicate()
score_changed.emit(score.duplicate())