mirror of
https://github.com/jcreek/CosmicClash.git
synced 2026-09-10 16:04:04 +00:00
2325313ad2
An Opus subagent's adversarial review of Phase 3 found a critical, silent, permanent bug plus eight smaller real issues, all empirically verified with real two- and three-process runs: CRITICAL: InputJitterBuffer's 32-entry ring permanently bricked a player's input once the un-consumed backlog exceeded the ring's capacity - a fresh arrival would land in the exact slot consume() was still waiting on, and since both counters only ever advance, the gap never closed. Reproduced with a real SIGSTOP/SIGCONT host freeze: client movement dropped from ~26m to 0.00m at ~0.7s, worse under real loss (a lossy link lowered the fatal threshold to ~400ms), and reachable via ordinary clock drift with no external trigger at all. Fixed by tracking the highest seq ever ingested and having consume() jump directly to what the ring can still provide once the gap exceeds capacity, instead of starving through an unrecoverable span. Re-verified with a 3s freeze (well past the original threshold): full recovery. HIGH: InputLeadController's release logic was gated on its own past attacks (lead > LEAD_MIN) rather than the real server-reported depth, so a backlog it didn't itself cause was never drained. Fixed to gate on actual depth vs target. MEDIUM-HIGH: the rate limiter's "N consecutive over-budget seconds" streak hard-reset to 0 on any clean window, letting a duty-cycled flood (burst, one clean window, repeat) sustain ~33x budget indefinitely with zero warnings. Replaced with a leaky-bucket accumulator immune to the same evasion by construction. MEDIUM: the seq > server_tick + 20 guard compared two unrelated clock epochs (server process uptime vs. client's own from-zero seq numbering), so it never actually protected anything on a long-running server and could silently drop an honest client's input forever. Bound against the buffer's own last_applied_seq instead. MEDIUM: InputJitterBuffer.stalled was computed but never reached the wire - the one signal that would have made the ring-overflow bug visible anywhere. Now wired through _ship_to_net_body_state. MEDIUM: task 3.6's CI driver's assertions didn't depend on client input reaching the server at all, so it kept passing with the ring-overflow bug actively triggered. Added real ship-movement and non-stalled checks, sampled while bots are still connected (an initial attempt sampled after their own legitimate disconnect, which starves identically to the bug). LOW-MEDIUM: a lead change silently mislabelled _input_history's older entries, since the wire format has no per-entry seq field. Fixed by handling each delta case (ordinary/release/attack) on its own terms. LOW: bandwidth and snapshot-loss overlay metrics froze at their last value during a total outage instead of decaying - exactly when they matter most. Both now report honest post-outage values. LOW: a guard comment on NetworkManager._ping misdescribed the actual disconnect_peer() arguments in use. Corrected. New permanent regression tests: test_ring_overflow_resyncs_to_fresh_data _instead_of_starving_forever, test_release_drains_a_backlog_it_never_ caused_itself, and client-abuse-flood-dutycycle (reproduces the exact duty-cycle evasion). Full regression suite, including the net-sim-latency milestone gate, all abuse roles, and the CI driver, re-run clean after every fix.
256 lines
12 KiB
GDScript
256 lines
12 KiB
GDScript
extends Node
|
|
|
|
# Autoload (project.godot [autoload] MatchSim). Phase 2 simulation RPCs:
|
|
# match_config (server assigns arena + deterministic slot order from
|
|
# MatchNet.roster), input (client -> server, per-tick action), snapshot
|
|
# (server -> client, NetCodec-packed body state), and a small score_update
|
|
# for the HUD. Lives on an autoload per §1.3's derived decision ("All
|
|
# hot-path RPCs live on autoloads") even though these are scoped to
|
|
# whichever match happens to be running — a scene-node RPC target would
|
|
# need matching NodePaths across peers, which an autoload sidesteps
|
|
# entirely, and it's what lets NetworkedMatch itself stay a plain scene
|
|
# node with no networking-identity concerns of its own.
|
|
#
|
|
# Channel intent per §2.1: 0 reliable (match_config, score_update), 1
|
|
# unreliable-ordered (input), 2 unreliable-ordered (snapshot) — not yet
|
|
# verified against ENet's own reserved system channel offset (§2.1's own
|
|
# "verify empirically" hedge); if that turns out to matter these indices
|
|
# will need adjusting, not the RPC design itself.
|
|
|
|
const NetCodec = preload("res://scripts/net_codec.gd")
|
|
|
|
signal match_config_received(arena_path: String, peer_ids: PackedInt32Array, teams: PackedInt32Array, spawn_indices: PackedInt32Array)
|
|
signal input_received(peer_id: int, decoded: Dictionary) # decoded: see NetCodec.unpack_input
|
|
signal snapshot_received(decoded: Dictionary) # decoded: see NetCodec.unpack_snapshot
|
|
signal score_update_received(score: Dictionary)
|
|
|
|
# Input validation (multiplayer-todo.md §3.1 steps 2-3, task 3.4). Deliberately
|
|
# lives here rather than in NetworkedMatch: framing/rate abuse is a protocol-
|
|
# level concern independent of any particular match's roster/slot state, and
|
|
# this autoload already owns the RPC that receives the raw bytes.
|
|
#
|
|
# 60Hz * 1.5 + 20, per §3.1 step 2's own numbers.
|
|
const RATE_LIMIT_PACKETS_PER_SEC := 110
|
|
# "Same for a byte budget" (§3.1 step 2) — the worst-case legitimate packet
|
|
# is a full-redundancy input (INPUT_HEADER_SIZE + MAX_REDUNDANCY entries,
|
|
# the "40 B input" §2.3 sizes to), so the byte budget is just the packet
|
|
# budget scaled by that worst-case size — no separate constant to keep in
|
|
# sync by hand.
|
|
const RATE_LIMIT_BYTES_PER_SEC := RATE_LIMIT_PACKETS_PER_SEC * (NetCodec.INPUT_HEADER_SIZE + NetCodec.MAX_REDUNDANCY * NetCodec.INPUT_ENTRY_SIZE)
|
|
const RATE_LIMIT_WINDOW_MS := 1000
|
|
# Leaky-bucket excess tolerance, expressed in the same "N seconds' worth of
|
|
# budget" terms the original consecutive-streak design used. An adversarial
|
|
# review found that design — a streak counter that HARD-RESET to 0 on any
|
|
# single clean window — was trivially evaded by a duty-cycled flood (burst,
|
|
# then one clean window, repeat): reproduced sustaining ~33x the packet
|
|
# budget indefinitely with zero disconnect warnings. A leaky bucket doesn't
|
|
# care how the excess is distributed in time — see the window-roll logic
|
|
# below for how it accumulates and drains.
|
|
const RATE_LIMIT_EXCESS_PACKETS_TO_DISCONNECT := RATE_LIMIT_PACKETS_PER_SEC * 3
|
|
const RATE_LIMIT_EXCESS_BYTES_TO_DISCONNECT := RATE_LIMIT_BYTES_PER_SEC * 3
|
|
const MALFORMED_LIMIT_TO_DISCONNECT := 20
|
|
|
|
|
|
class _PeerInputState:
|
|
var window_start_ms := 0
|
|
var packets_this_window := 0
|
|
var bytes_this_window := 0
|
|
# Leaky bucket: grows by this window's actual total, drains by one
|
|
# window's worth of budget, every window — regardless of whether that
|
|
# window was itself over or under budget. A steady rate at or under
|
|
# budget nets to zero forever (never accumulates); any sustained AVERAGE
|
|
# above budget accumulates over time no matter how it's shaped into
|
|
# bursts, unlike a streak counter a clean gap can reset to 0.
|
|
var excess_packets := 0.0
|
|
var excess_bytes := 0.0
|
|
var malformed_count := 0
|
|
|
|
|
|
var _peer_input_state: Dictionary = {} # peer_id -> _PeerInputState, server only
|
|
|
|
# Bandwidth (task 3.7's debug overlay): only the two 60Hz hot-path channels
|
|
# (input, snapshot) — match_config/score_update are low-frequency control
|
|
# messages, not what §2's byte-budget analysis or a live overlay cares
|
|
# about. Rolling per-second counters, recomputed opportunistically on each
|
|
# send/receive rather than on a timer — nothing needs the rate outside of
|
|
# an on-demand overlay read anyway. Use get_bytes_sent_per_sec() /
|
|
# get_bytes_received_per_sec() to READ these, not the raw fields directly
|
|
# — see those functions for why.
|
|
const BANDWIDTH_WINDOW_MS := 1000
|
|
var bytes_sent_per_sec := 0.0
|
|
var bytes_received_per_sec := 0.0
|
|
var _sent_window_start_ms := 0
|
|
var _sent_window_bytes := 0
|
|
var _received_window_start_ms := 0
|
|
var _received_window_bytes := 0
|
|
|
|
|
|
# An adversarial review found bytes_*_per_sec only ever gets recomputed
|
|
# INSIDE _track_sent()/_track_received() — i.e. only when traffic actually
|
|
# arrives — so if traffic stops entirely (right before a disconnect, or
|
|
# during exactly the kind of outage this overlay exists to diagnose), the
|
|
# last computed rate displays forever instead of decaying toward zero.
|
|
# Report zero once meaningfully more than one window has passed with
|
|
# nothing tracked, rather than trusting a stale field.
|
|
func get_bytes_sent_per_sec() -> float:
|
|
if Time.get_ticks_msec() - _sent_window_start_ms > BANDWIDTH_WINDOW_MS * 2:
|
|
return 0.0
|
|
return bytes_sent_per_sec
|
|
|
|
|
|
func get_bytes_received_per_sec() -> float:
|
|
if Time.get_ticks_msec() - _received_window_start_ms > BANDWIDTH_WINDOW_MS * 2:
|
|
return 0.0
|
|
return bytes_received_per_sec
|
|
|
|
|
|
func _ready() -> void:
|
|
NetworkManager.client_disconnected.connect(func(peer_id: int) -> void: _peer_input_state.erase(peer_id))
|
|
|
|
|
|
func _track_sent(n: int) -> void:
|
|
var now := Time.get_ticks_msec()
|
|
if now - _sent_window_start_ms >= BANDWIDTH_WINDOW_MS:
|
|
bytes_sent_per_sec = _sent_window_bytes * 1000.0 / maxf(1.0, float(now - _sent_window_start_ms))
|
|
_sent_window_start_ms = now
|
|
_sent_window_bytes = 0
|
|
_sent_window_bytes += n
|
|
|
|
|
|
func _track_received(n: int) -> void:
|
|
var now := Time.get_ticks_msec()
|
|
if now - _received_window_start_ms >= BANDWIDTH_WINDOW_MS:
|
|
bytes_received_per_sec = _received_window_bytes * 1000.0 / maxf(1.0, float(now - _received_window_start_ms))
|
|
_received_window_start_ms = now
|
|
_received_window_bytes = 0
|
|
_received_window_bytes += n
|
|
|
|
# Server only: the last match_config actually sent, so a client whose own
|
|
# scene load (and therefore its match_config_received listener) finishes
|
|
# AFTER the server already broadcast can still get it — a one-shot
|
|
# broadcast alone is racy against however long the client takes to reach
|
|
# the point where it's listening, and Godot signals never buffer for a
|
|
# late connection. request_match_config() closes that race by turning
|
|
# delivery into "ask until you get it" instead of "hope you were already
|
|
# listening." Also covers a late joiner mid-match (Phase 5 will still need
|
|
# to add live match *state*, not just this static config, for that case).
|
|
var _last_match_config: Dictionary = {}
|
|
|
|
|
|
func send_match_config(arena_path: String, peer_ids: PackedInt32Array, teams: PackedInt32Array, spawn_indices: PackedInt32Array) -> void:
|
|
_last_match_config = {
|
|
"arena_path": arena_path, "peer_ids": peer_ids, "teams": teams, "spawn_indices": spawn_indices,
|
|
}
|
|
_match_config.rpc(arena_path, peer_ids, teams, spawn_indices)
|
|
|
|
|
|
func request_match_config() -> void:
|
|
_request_match_config.rpc_id(1)
|
|
|
|
|
|
func send_input(bytes: PackedByteArray) -> void:
|
|
_track_sent(bytes.size())
|
|
# bytes is already fully packed (any timestamps it carries are already
|
|
# fixed), so wrapping the dispatch itself is enough — task 2.8.
|
|
NetSim.send(func() -> void: _recv_input.rpc_id(1, bytes), 1)
|
|
|
|
|
|
func send_snapshot(peer_id: int, bytes: PackedByteArray) -> void:
|
|
_track_sent(bytes.size())
|
|
NetSim.send(func() -> void: _snapshot.rpc_id(peer_id, bytes), peer_id)
|
|
|
|
|
|
func send_score_update(score: Dictionary) -> void:
|
|
_score_update.rpc(score)
|
|
|
|
|
|
@rpc("authority", "call_remote", "reliable", 0)
|
|
func _match_config(arena_path: String, peer_ids: PackedInt32Array, teams: PackedInt32Array, spawn_indices: PackedInt32Array) -> void:
|
|
match_config_received.emit(arena_path, peer_ids, teams, spawn_indices)
|
|
|
|
|
|
@rpc("any_peer", "call_remote", "reliable", 0)
|
|
func _request_match_config() -> void:
|
|
if not multiplayer.is_server() or _last_match_config.is_empty():
|
|
return
|
|
var peer_id := multiplayer.get_remote_sender_id()
|
|
_match_config.rpc_id(
|
|
peer_id, _last_match_config["arena_path"], _last_match_config["peer_ids"],
|
|
_last_match_config["teams"], _last_match_config["spawn_indices"]
|
|
)
|
|
|
|
|
|
@rpc("any_peer", "call_remote", "unreliable_ordered", 1)
|
|
func _recv_input(bytes: PackedByteArray) -> void:
|
|
if not multiplayer.is_server():
|
|
return
|
|
_track_received(bytes.size())
|
|
var peer_id := multiplayer.get_remote_sender_id()
|
|
|
|
var state: _PeerInputState = _peer_input_state.get(peer_id)
|
|
if state == null:
|
|
state = _PeerInputState.new()
|
|
_peer_input_state[peer_id] = state
|
|
|
|
# Rolling 1s window (§3.1 step 2). Rolled over lazily on the first
|
|
# packet past the window boundary, not on a timer — this RPC only ever
|
|
# runs when a packet actually arrives, so there's nothing to roll over
|
|
# when nothing is arriving anyway.
|
|
var now_ms := Time.get_ticks_msec()
|
|
if now_ms - state.window_start_ms >= RATE_LIMIT_WINDOW_MS:
|
|
state.excess_packets = maxf(0.0, state.excess_packets + float(state.packets_this_window) - float(RATE_LIMIT_PACKETS_PER_SEC))
|
|
state.excess_bytes = maxf(0.0, state.excess_bytes + float(state.bytes_this_window) - float(RATE_LIMIT_BYTES_PER_SEC))
|
|
state.window_start_ms = now_ms
|
|
state.packets_this_window = 0
|
|
state.bytes_this_window = 0
|
|
if state.excess_packets > RATE_LIMIT_EXCESS_PACKETS_TO_DISCONNECT or state.excess_bytes > RATE_LIMIT_EXCESS_BYTES_TO_DISCONNECT:
|
|
_disconnect_abusive_peer(peer_id, "input rate limit exceeded (excess_packets=%.0f excess_bytes=%.0f)" % [state.excess_packets, state.excess_bytes])
|
|
return
|
|
|
|
state.packets_this_window += 1
|
|
state.bytes_this_window += bytes.size()
|
|
if state.packets_this_window > RATE_LIMIT_PACKETS_PER_SEC or state.bytes_this_window > RATE_LIMIT_BYTES_PER_SEC:
|
|
return # over budget for the current window — drop, counted above at the next window roll
|
|
|
|
# Framing (§3.1 step 3), validated before decoding — unpack_input can't
|
|
# be trusted to catch this itself: StreamPeerBuffer silently zero-fills
|
|
# past EOF rather than erroring (found during Phase 2's adversarial
|
|
# review's hostile-client stress test), so a too-short or size-mismatched
|
|
# payload would otherwise decode "successfully" into garbage actions
|
|
# instead of being rejected.
|
|
if bytes.size() < NetCodec.INPUT_HEADER_SIZE:
|
|
_count_malformed(peer_id, state)
|
|
return
|
|
var count: int = bytes[5] # type_version(1) + seq(4) precede count — see pack_input's own layout
|
|
if count == 0 or count > NetCodec.MAX_REDUNDANCY or bytes.size() != NetCodec.INPUT_HEADER_SIZE + count * NetCodec.INPUT_ENTRY_SIZE:
|
|
_count_malformed(peer_id, state)
|
|
return
|
|
|
|
var decoded := NetCodec.unpack_input(bytes)
|
|
input_received.emit(peer_id, decoded)
|
|
|
|
|
|
func _count_malformed(peer_id: int, state: _PeerInputState) -> void:
|
|
state.malformed_count += 1
|
|
if state.malformed_count >= MALFORMED_LIMIT_TO_DISCONNECT:
|
|
_disconnect_abusive_peer(peer_id, "too many malformed input packets (%d)" % state.malformed_count)
|
|
|
|
|
|
func _disconnect_abusive_peer(peer_id: int, reason: String) -> void:
|
|
push_warning("MatchSim: disconnecting peer %d for abuse: %s" % [peer_id, reason])
|
|
_peer_input_state.erase(peer_id)
|
|
if multiplayer.multiplayer_peer is ENetMultiplayerPeer:
|
|
multiplayer.multiplayer_peer.disconnect_peer(peer_id)
|
|
|
|
|
|
@rpc("authority", "call_remote", "unreliable_ordered", 2)
|
|
func _snapshot(bytes: PackedByteArray) -> void:
|
|
_track_received(bytes.size())
|
|
var decoded := NetCodec.unpack_snapshot(bytes)
|
|
snapshot_received.emit(decoded)
|
|
|
|
|
|
@rpc("authority", "call_remote", "reliable", 0)
|
|
func _score_update(score: Dictionary) -> void:
|
|
score_update_received.emit(score)
|