feat(training): add opt-in teamplay evaluation

This commit is contained in:
Josh Creek
2026-09-01 17:30:40 +01:00
parent 9004800326
commit 7b2f9c26f4
8 changed files with 84 additions and 11 deletions
+17 -1
View File
@@ -38,6 +38,10 @@ extends AIController3D
# stone, matching ball_touch_cooldown_ticks's existing "stepping-stone, not # stone, matching ball_touch_cooldown_ticks's existing "stepping-stone, not
# the objective" framing. # the objective" framing.
@export_range(0.0, 1.0) var ball_touch_direction_floor := 0.3 @export_range(0.0, 1.0) var ball_touch_direction_floor := 0.3
# Fraction of a touch payout shared with teammates. Zero preserves all
# existing 1v1/curriculum reward functions; in teamplay the shared amount is
# divided across teammates and never exceeds the touching ship's payout.
@export_range(0.0, 1.0) var team_touch_credit_weight := 0.0
@export var velocity_to_ball_weight := 0.02 @export var velocity_to_ball_weight := 0.02
# Dense reward for approaching the ball *nose first* near the floor. Unlike # Dense reward for approaching the ball *nose first* near the floor. Unlike
# velocity_to_ball_weight, sideways/reverse closing velocity earns nothing: # velocity_to_ball_weight, sideways/reverse closing velocity earns nothing:
@@ -606,6 +610,12 @@ func _on_ship_body_entered(body: Node) -> void:
if air_touch_bonus_weight > 0.0 and ball.global_position.y > AIR_TOUCH_HEIGHT: if air_touch_bonus_weight > 0.0 and ball.global_position.y > AIR_TOUCH_HEIGHT:
touch_payout += air_touch_bonus_weight * alignment touch_payout += air_touch_bonus_weight * alignment
reward += touch_payout reward += touch_payout
if team_touch_credit_weight > 0.0 and not teammates.is_empty():
var teammate_credit := team_touch_credit(touch_payout, team_touch_credit_weight, teammates.size())
for teammate in teammates:
var teammate_agent := teammate.get_node_or_null("ShipAIController") as ShipAIController
if is_instance_valid(teammate_agent):
teammate_agent.reward += teammate_credit
_ticks_since_ball_touch = 0 _ticks_since_ball_touch = 0
# air_touch_fraction/productive_air_touch_fraction (see get_info) share # air_touch_fraction/productive_air_touch_fraction (see get_info) share
@@ -616,4 +626,10 @@ func _on_ship_body_entered(body: Node) -> void:
if ball.global_position.y > AIR_TOUCH_HEIGHT: if ball.global_position.y > AIR_TOUCH_HEIGHT:
_air_touches += 1 _air_touches += 1
if alignment >= PRODUCTIVE_AIR_TOUCH_ALIGNMENT: if alignment >= PRODUCTIVE_AIR_TOUCH_ALIGNMENT:
_productive_air_touches += 1 _productive_air_touches += 1
static func team_touch_credit(touch_payout: float, weight: float, teammate_count: int) -> float:
if touch_payout <= 0.0 or weight <= 0.0 or teammate_count <= 0:
return 0.0
return touch_payout * clampf(weight, 0.0, 1.0) / teammate_count
+10 -6
View File
@@ -141,6 +141,7 @@ var _eval_goals := {0: 0, 1: 0}
var _eval_draws := 0 var _eval_draws := 0
var _eval_episodes_done := 0 var _eval_episodes_done := 0
var _episode_ticks := 0 var _episode_ticks := 0
var _eval_team_size := 1
# Curriculum mode state (see _parse_curriculum_args). "self_play" (default) # Curriculum mode state (see _parse_curriculum_args). "self_play" (default)
# is today's only historical behaviour: both ships are live trainees sharing # is today's only historical behaviour: both ships are live trainees sharing
@@ -176,11 +177,12 @@ func _start() -> void:
spawn_ball() spawn_ball()
if _eval: if _eval:
for team in [0, 1]: for team in [0, 1]:
var bot := AIShipController.new() for spawn_index in _eval_team_size:
bot.model_path = _eval_models[team] var bot := AIShipController.new()
bot.allow_vertical = _eval_allow_vertical[team] bot.model_path = _eval_models[team]
bot.allow_pitch_roll = _eval_allow_pitch_roll[team] bot.allow_vertical = _eval_allow_vertical[team]
spawn_ship(team, 0, bot) bot.allow_pitch_roll = _eval_allow_pitch_roll[team]
spawn_ship(team, spawn_index, bot)
return return
var team0_ships: Array[Ship] = [] var team0_ships: Array[Ship] = []
@@ -248,6 +250,7 @@ func _parse_eval_args() -> void:
_eval_models[0] = args["eval_model_a"] _eval_models[0] = args["eval_model_a"]
_eval_models[1] = args["eval_model_b"] _eval_models[1] = args["eval_model_b"]
_eval_episodes = int(args.get("eval_episodes", str(_eval_episodes))) _eval_episodes = int(args.get("eval_episodes", str(_eval_episodes)))
_eval_team_size = clampi(int(args.get("eval_team_size", str(_eval_team_size))), 1, 2)
_eval_allow_vertical[0] = _typed_like(args.get("eval_allow_vertical_a", "true"), true) _eval_allow_vertical[0] = _typed_like(args.get("eval_allow_vertical_a", "true"), true)
_eval_allow_vertical[1] = _typed_like(args.get("eval_allow_vertical_b", "true"), true) _eval_allow_vertical[1] = _typed_like(args.get("eval_allow_vertical_b", "true"), true)
_eval_allow_pitch_roll[0] = _typed_like(args.get("eval_allow_pitch_roll_a", "true"), true) _eval_allow_pitch_roll[0] = _typed_like(args.get("eval_allow_pitch_roll_a", "true"), true)
@@ -265,7 +268,7 @@ const TRAINING_MODE_OVERRIDES := [
# ShipAIController @export names a curriculum run may override, read as # ShipAIController @export names a curriculum run may override, read as
# --ai_<name>=<value> to avoid colliding with the names above. # --ai_<name>=<value> to avoid colliding with the names above.
const SHIP_AI_OVERRIDES := [ const SHIP_AI_OVERRIDES := [
"ball_touch_reward", "ball_touch_cooldown_ticks", "ball_touch_direction_floor", "ball_touch_reward", "ball_touch_cooldown_ticks", "ball_touch_direction_floor", "team_touch_credit_weight",
"velocity_to_ball_weight", "ball_velocity_to_goal_weight", "ball_distance_penalty", "velocity_to_ball_weight", "ball_velocity_to_goal_weight", "ball_distance_penalty",
"forward_velocity_to_ball_weight", "air_approach_weight", "air_touch_bonus_weight", "wall_contact_penalty", "tilt_penalty", "forward_velocity_to_ball_weight", "air_approach_weight", "air_touch_bonus_weight", "wall_contact_penalty", "tilt_penalty",
"ground_tilt_penalty", "non_forward_penalty", "grounded_upright_reward", "ground_tilt_penalty", "non_forward_penalty", "grounded_upright_reward",
@@ -323,6 +326,7 @@ func _ai_default(name: String) -> Variant:
"ball_touch_reward": return 0.4 "ball_touch_reward": return 0.4
"ball_touch_cooldown_ticks": return 60 "ball_touch_cooldown_ticks": return 60
"ball_touch_direction_floor": return 0.3 "ball_touch_direction_floor": return 0.3
"team_touch_credit_weight": return 0.0
"velocity_to_ball_weight": return 0.02 "velocity_to_ball_weight": return 0.02
"forward_velocity_to_ball_weight": return 0.0 "forward_velocity_to_ball_weight": return 0.0
"air_approach_weight": return 0.0 "air_approach_weight": return 0.0
+18
View File
@@ -0,0 +1,18 @@
extends TestCase
const ShipAIControllerScript = preload("res://scripts/ship_ai_controller.gd")
func test_team_touch_credit_is_split_across_teammates() -> void:
assert_eq(ShipAIControllerScript.team_touch_credit(0.8, 0.5, 2), 0.2, "half-weight touch is split between two teammates")
func test_team_touch_credit_never_exceeds_touch_payout() -> void:
var credit := ShipAIControllerScript.team_touch_credit(0.8, 1.0, 1)
assert_eq(credit, 0.8, "one teammate receives at most the touch payout")
func test_team_touch_credit_rejects_invalid_inputs() -> void:
assert_eq(ShipAIControllerScript.team_touch_credit(0.8, 0.0, 2), 0.0, "zero weight is disabled")
assert_eq(ShipAIControllerScript.team_touch_credit(0.8, 0.5, 0), 0.0, "no teammates receive no credit")
assert_eq(ShipAIControllerScript.team_touch_credit(-1.0, 0.5, 2), 0.0, "negative payout cannot mint reward")
func test_team_touch_credit_clamps_weight() -> void:
assert_eq(ShipAIControllerScript.team_touch_credit(0.8, 2.0, 2), 0.4, "weight above one is clamped")
+1 -1
View File
@@ -8,7 +8,7 @@ The training pipeline is built — see `TRAINING.md` (self-play PPO via the vend
- [ ] Run the generation-5 handling/intercepts/league/teamplay curriculum described in `TRAINING.md`; promote later checkpoints as `medium`/`hard` only after they clear the match and behaviour gates. The orchestrator now requires three independent paired evaluation seeds for each promotion decision; the current Stage 6 league run remains blocked on its recorded regression/telemetry results. - [ ] Run the generation-5 handling/intercepts/league/teamplay curriculum described in `TRAINING.md`; promote later checkpoints as `medium`/`hard` only after they clear the match and behaviour gates. The orchestrator now requires three independent paired evaluation seeds for each promotion decision; the current Stage 6 league run remains blocked on its recorded regression/telemetry results.
- [ ] Extend generation 5's moving aerial-intercept states with wall plays and rebound scenarios after Stage 5 establishes a productive-air-touch baseline. - [ ] Extend generation 5's moving aerial-intercept states with wall plays and rebound scenarios after Stage 5 establishes a productive-air-touch baseline.
- [ ] Design team-credit rewards and paired 2v2 evaluation before enabling the deferred teamplay stage. - [x] Design team-credit rewards and paired 2v2 evaluation before enabling the deferred teamplay stage. `team_touch_credit_weight` is zero by default and `evaluate.py --team-size=2` provides the opt-in paired evaluator; Stage 7 remains disabled pending recorded 2v2 behaviour gates.
## Presentation / AAA polish ## Presentation / AAA polish
+6 -1
View File
@@ -667,7 +667,12 @@ can be based on evidence instead of a single watched match.
Stage 7 teamplay remains deliberately unconfigured. The fixed roster Stage 7 teamplay remains deliberately unconfigured. The fixed roster
observation and `team_size` plumbing can run 2v2, but there is no paired 2v2 observation and `team_size` plumbing can run 2v2, but there is no paired 2v2
evaluation or team-credit reward yet; spending 120M steps without those gates evaluation or team-credit reward yet; spending 120M steps without those gates
would make a pass meaningless. would make a pass meaningless. The prerequisites are now implemented but
remain opt-in: `ShipAIController.team_touch_credit_weight` shares a bounded
fraction of a touch payout across same-team agents (default `0.0` preserves
all existing curricula), and `evaluate.py --team-size=2` runs the same policy
as a two-ship team with the existing paired side swap. Stage 7 stays
unconfigured until a recorded 2v2 evaluation establishes teamplay gates.
`training/generation5.py` implements Stages 46 separately from the completed `training/generation5.py` implements Stages 46 separately from the completed
generation-4 orchestrator and state. It always begins Stage 4 from generation-4 orchestrator and state. It always begins Stage 4 from
+6
View File
@@ -1457,6 +1457,12 @@ The following Phase 8 slices have local implementation and verification evidence
The following are not locally certifiable from this workspace and remain open prerequisites rather than silently “done”: Valve/GodotSteam credentials and hosted SDR (7.17.8), live PostgreSQL/Redis execution where Docker is unavailable, live Agones/kind lifecycle (8.308.38, 8.49), public-network chaos/load/cost/release gates (8.508.53), and real-hardware graphics profiling (0.15b onward). `TODO.md`s AI-training and presentation tasks remain separate from multiplayer and are not marked by this index. The following are not locally certifiable from this workspace and remain open prerequisites rather than silently “done”: Valve/GodotSteam credentials and hosted SDR (7.17.8), live PostgreSQL/Redis execution where Docker is unavailable, live Agones/kind lifecycle (8.308.38, 8.49), public-network chaos/load/cost/release gates (8.508.53), and real-hardware graphics profiling (0.15b onward). `TODO.md`s AI-training and presentation tasks remain separate from multiplayer and are not marked by this index.
The deferred teamplay TODO prerequisite is now implemented locally but not
enabled: team-touch credit is opt-in and the evaluator can run paired 2v2
matches with `--team-size=2`. No Stage 7 training run or promotion is claimed;
teamplay still needs recorded behaviour thresholds and a working Godot runtime
for its end-to-end evaluation.
The three declared domain fuzz targets have now each completed a bounded 4-second run (`FuzzQueueCreateDoesNotPanic`, `FuzzResultDigestIsDeterministic`, and `FuzzSyncEventApplicationDoesNotPanic`) with no failures; this closes the locally runnable fuzz portion of 8.46. Live Redis failover, further transaction races, and cloud/runtime gates remain explicitly unverified. The three declared domain fuzz targets have now each completed a bounded 4-second run (`FuzzQueueCreateDoesNotPanic`, `FuzzResultDigestIsDeterministic`, and `FuzzSyncEventApplicationDoesNotPanic`) with no failures; this closes the locally runnable fuzz portion of 8.46. Live Redis failover, further transaction races, and cloud/runtime gates remain explicitly unverified.
The real ENet integration gate now passes with `GODOT_BIN=/Applications/Godot.app/Contents/MacOS/Godot bash scripts/verify_enet_integration.sh`, covering the `net`, `match-net`, `clock`, `lobby`, and `networked match` process scenarios. The default `GODOT_BIN` remains the portable `godot` PATH lookup for CI; this machine requires the explicit app-bundle path. The real ENet integration gate now passes with `GODOT_BIN=/Applications/Godot.app/Contents/MacOS/Godot bash scripts/verify_enet_integration.sh`, covering the `net`, `match-net`, `clock`, `lobby`, and `networked match` process scenarios. The default `GODOT_BIN` remains the portable `godot` PATH lookup for CI; this machine requires the explicit app-bundle path.
+10
View File
@@ -33,6 +33,7 @@ def run_half(
seed: int, seed: int,
grounded_a: bool = False, grounded_a: bool = False,
grounded_b: bool = False, grounded_b: bool = False,
team_size: int = 1,
) -> dict: ) -> dict:
cmd = [ cmd = [
godot_bin, godot_bin,
@@ -47,6 +48,10 @@ def run_half(
f"--speedup={speedup}", f"--speedup={speedup}",
f"--env_seed={seed}", f"--env_seed={seed}",
] ]
if team_size not in (1, 2):
raise ValueError("team_size must be 1 or 2")
if team_size == 2:
cmd.append("--eval_team_size=2")
# Must match how each model was actually trained (see AIShipController's # Must match how each model was actually trained (see AIShipController's
# allow_vertical/allow_pitch_roll) — a grounded pre-generation-4 model # allow_vertical/allow_pitch_roll) — a grounded pre-generation-4 model
# never got a reward gradient on these axes, so leaving them unmasked here # never got a reward gradient on these axes, so leaving them unmasked here
@@ -72,6 +77,7 @@ def evaluate_pair(
seed: int, seed: int,
grounded_a: bool = False, grounded_a: bool = False,
grounded_b: bool = False, grounded_b: bool = False,
team_size: int = 1,
) -> dict: ) -> dict:
"""Replay one seeded state sequence with the models on opposite sides.""" """Replay one seeded state sequence with the models on opposite sides."""
if episodes < 2 or episodes % 2 != 0: if episodes < 2 or episodes % 2 != 0:
@@ -81,10 +87,12 @@ def evaluate_pair(
first = run_half( first = run_half(
godot_bin, model_a, model_b, episodes_per_side, speedup, seed, godot_bin, model_a, model_b, episodes_per_side, speedup, seed,
grounded_a=grounded_a, grounded_b=grounded_b, grounded_a=grounded_a, grounded_b=grounded_b,
team_size=team_size,
) )
second = run_half( second = run_half(
godot_bin, model_b, model_a, episodes_per_side, speedup, seed, godot_bin, model_b, model_a, episodes_per_side, speedup, seed,
grounded_a=grounded_b, grounded_b=grounded_a, grounded_a=grounded_b, grounded_b=grounded_a,
team_size=team_size,
) )
a_team_0 = { a_team_0 = {
@@ -133,6 +141,7 @@ def main():
help="Path to the Godot binary (or set GODOT_BIN)", help="Path to the Godot binary (or set GODOT_BIN)",
) )
parser.add_argument("--speedup", type=int, default=16) parser.add_argument("--speedup", type=int, default=16)
parser.add_argument("--team-size", type=int, choices=(1, 2), default=1)
parser.add_argument("--seed", type=int, default=1, help="Seed for the paired starting-state sequence") parser.add_argument("--seed", type=int, default=1, help="Seed for the paired starting-state sequence")
parser.add_argument("--history", default=str(TRAINING_DIR / "eval_history.json")) parser.add_argument("--history", default=str(TRAINING_DIR / "eval_history.json"))
parser.add_argument( parser.add_argument(
@@ -149,6 +158,7 @@ def main():
record = evaluate_pair( record = evaluate_pair(
args.godot_bin, model_a, model_b, args.episodes, args.speedup, args.seed, args.godot_bin, model_a, model_b, args.episodes, args.speedup, args.seed,
grounded_a=args.grounded_a, grounded_b=args.grounded_b, grounded_a=args.grounded_a, grounded_b=args.grounded_b,
team_size=args.team_size,
) )
except ValueError as error: except ValueError as error:
parser.error(str(error)) parser.error(str(error))
+16 -2
View File
@@ -23,11 +23,11 @@ class EvaluatePairTests(unittest.TestCase):
self.assertEqual(run_half.call_args_list[1].args, ("godot", "reference.json", "candidate.json", 4, 16, 42)) self.assertEqual(run_half.call_args_list[1].args, ("godot", "reference.json", "candidate.json", 4, 16, 42))
self.assertEqual( self.assertEqual(
run_half.call_args_list[0].kwargs, run_half.call_args_list[0].kwargs,
{"grounded_a": False, "grounded_b": True}, {"grounded_a": False, "grounded_b": True, "team_size": 1},
) )
self.assertEqual( self.assertEqual(
run_half.call_args_list[1].kwargs, run_half.call_args_list[1].kwargs,
{"grounded_a": True, "grounded_b": False}, {"grounded_a": True, "grounded_b": False, "team_size": 1},
) )
self.assertEqual(record["wins_a"], 4) self.assertEqual(record["wins_a"], 4)
self.assertEqual(record["wins_b"], 3) self.assertEqual(record["wins_b"], 3)
@@ -42,6 +42,20 @@ class EvaluatePairTests(unittest.TestCase):
with self.assertRaisesRegex(ValueError, "even number"): with self.assertRaisesRegex(ValueError, "even number"):
evaluate.evaluate_pair("godot", "a", "b", episodes, 16, 1) evaluate.evaluate_pair("godot", "a", "b", episodes, 16, 1)
def test_rejects_unsupported_team_size_before_launch(self) -> None:
with self.assertRaisesRegex(ValueError, "team_size must be 1 or 2"):
evaluate.run_half("godot", "a", "b", 2, 16, 1, team_size=3)
@patch("evaluate.run_half")
def test_2v2_evaluation_preserves_side_swap_and_team_size(self, run_half) -> None:
run_half.side_effect = [
{"episodes": 2, "goals_a": 1, "goals_b": 0, "draws": 1},
{"episodes": 2, "goals_a": 0, "goals_b": 1, "draws": 1},
]
evaluate.evaluate_pair("godot", "a", "b", 4, 16, 9, team_size=2)
self.assertEqual(run_half.call_args_list[0].kwargs["team_size"], 2)
self.assertEqual(run_half.call_args_list[1].kwargs["team_size"], 2)
@patch("evaluate.run_half") @patch("evaluate.run_half")
def test_identical_policy_results_cancel_physical_side_bias(self, run_half) -> None: def test_identical_policy_results_cancel_physical_side_bias(self, run_half) -> None:
# Replaying the same deterministic matchup must produce the same # Replaying the same deterministic matchup must produce the same