From 6f7536f03c6ed312670e78d7e3dd00e398d0d615 Mon Sep 17 00:00:00 2001 From: Josh Creek <8179928+jcreek@users.noreply.github.com> Date: Sun, 9 Aug 2026 13:23:00 +0100 Subject: [PATCH] fix(training): correct non-forward penalty math and add a grounding incentive Adversarial review of the previous stage-4 retune found two problems: non_forward_speed used planar_speed - forward_component, which under-charges diagonal motion relative to true lateral speed (e.g. ~29% penalty at 45 degrees off the nose instead of the correct ~71%); fixed to the Pythagorean magnitude for forward-facing angles, full speed for backward-facing ones. Also, ground_tilt_penalty and non_forward_penalty only ever cost reward near the floor with nothing offsetting them above it, which could teach a policy that's still bad at ground handling to just avoid the floor rather than get better at it. Added grounded_upright_reward (ship_ai_controller.gd) plus a new ShipObservations.is_floor_contact helper for genuine belly-on-floor contact detection, so grounding well while upright is the locally profitable choice, not just the least-punished one. --- Game/scripts/ship_ai_controller.gd | 38 ++++++++++++++++++++++++++++-- Game/scripts/ship_observations.gd | 18 ++++++++++++++ Game/scripts/training_mode.gd | 5 ++-- training/generation5.py | 9 ++++++- training/train.py | 6 +++++ 5 files changed, 71 insertions(+), 5 deletions(-) diff --git a/Game/scripts/ship_ai_controller.gd b/Game/scripts/ship_ai_controller.gd index 54cf00d8..2b296fc1 100644 --- a/Game/scripts/ship_ai_controller.gd +++ b/Game/scripts/ship_ai_controller.gd @@ -79,6 +79,20 @@ extends AIController3D # forward_velocity_to_ball_weight's ball-conditioned bonus. Same # GROUND_HANDLING_HEIGHT altitude fade as ground_tilt_penalty. @export var non_forward_penalty := 0.0 +# Per-tick bonus for genuinely resting on the floor (ShipObservations. +# is_floor_contact, real contact — not just being below +# GROUND_HANDLING_HEIGHT) while upright. The positive counterpart to +# ground_tilt_penalty/non_forward_penalty: without it, staying above +# GROUND_HANDLING_HEIGHT is reward-neutral relative to grounding, so a +# policy that's still bad at ground handling could "solve" those penalties +# by just avoiding the floor rather than by getting better at handling on +# it — worsening Stage 3's already-airborne-heavy baseline instead of +# fixing it. Kept an order of magnitude below ball_touch_reward/goal_reward +# and comparable to time_penalty/ball_distance_penalty so grounding well is +# attractive without making idling upright on the spot, away from the ball, +# competitive with actually playing (see ball_distance_penalty's run04 +# lesson on why a flat positional bonus needs a countervailing cost). +@export var grounded_upright_reward := 0.0 # Per-tick bonus for own speed: 0 stationary, full value (+0.24/s) at # max_speed. Run07 lesson: after the kickoff flurry both ships parked next to # a cornered ball — with every other dense term near zero there, standing @@ -333,16 +347,36 @@ func _physics_process(delta): # (sideways or reverse), independent of the ball — the mirror image of # forward_velocity_to_ball_weight's ball-conditioned bonus. Fades out with # altitude via the same GROUND_HANDLING_HEIGHT ramp as ground_tilt_penalty. + # non_forward_speed is the true lateral magnitude (Pythagorean, not the + # cruder planar_speed - forward_component, which under-charges diagonal + # motion — e.g. at 45 degrees off the nose that gave ~29% of full-speed + # penalty instead of the correct ~71%) for any forward-facing component; + # a backward-facing component (dot product below zero) is fully + # penalized regardless of angle, same as pure sideways motion. if non_forward_penalty > 0.0 and ship.global_position.y < GROUND_HANDLING_HEIGHT: var non_forward_planar_velocity := Vector3(ship.linear_velocity.x, 0.0, ship.linear_velocity.z) var non_forward_planar_speed := non_forward_planar_velocity.length() var non_forward_planar_forward := Vector3(-ship.global_transform.basis.z.x, 0.0, -ship.global_transform.basis.z.z) if non_forward_planar_speed > 0.0001 and non_forward_planar_forward.length_squared() > 0.0001: - var forward_component: float = maxf(non_forward_planar_velocity.dot(non_forward_planar_forward.normalized()), 0.0) - var non_forward_speed: float = non_forward_planar_speed - forward_component + var forward_component: float = non_forward_planar_velocity.dot(non_forward_planar_forward.normalized()) + var non_forward_speed: float + if forward_component >= 0.0: + non_forward_speed = sqrt(maxf( + non_forward_planar_speed * non_forward_planar_speed - forward_component * forward_component, 0.0 + )) + else: + non_forward_speed = non_forward_planar_speed var non_forward_ground_factor: float = 1.0 - clampf(ship.global_position.y / GROUND_HANDLING_HEIGHT, 0.0, 1.0) reward -= non_forward_penalty * (non_forward_speed / ship.max_speed) * non_forward_ground_factor + # Dense bonus: genuinely resting on the floor while upright (see + # grounded_upright_reward) — the positive counterpart to + # ground_tilt_penalty/non_forward_penalty, so grounding is worth + # pursuing, not just less punished than staying airborne. + if grounded_upright_reward > 0.0 and ShipObservations.is_floor_contact(ship): + var grounded_uprightness: float = ship.global_transform.basis.y.dot(Vector3.UP) + reward += grounded_upright_reward * maxf(grounded_uprightness, 0.0) + # Dense penalty: height above the floor (see airborne_penalty). The # floor sits at world y = 0 (see training_mode.gd's FIELD_MIN_Y/ # _escaped bounds); normalized so the worst case is pinned at the diff --git a/Game/scripts/ship_observations.gd b/Game/scripts/ship_observations.gd index 90035e1f..dbf8d367 100644 --- a/Game/scripts/ship_observations.gd +++ b/Game/scripts/ship_observations.gd @@ -143,3 +143,21 @@ static func contact_normal(ship: Ship) -> Vector3: if normal.y < FLOOR_NORMAL_MIN_Y: return normal return Vector3.ZERO + + +# True belly-on-floor contact — the complement of contact_normal, which +# deliberately excludes floor contact (see its comment). Used by +# ShipAIController.grounded_upright_reward to reward genuinely resting on +# the floor rather than just being below the GROUND_HANDLING_HEIGHT proxy +# altitude, so a ship can't collect ground-handling reward by hovering just +# under the threshold without ever touching down. +static func is_floor_contact(ship: Ship) -> bool: + var state := PhysicsServer3D.body_get_direct_state(ship.get_rid()) + if state == null: + return false + for i in state.get_contact_count(): + if not state.get_contact_collider_object(i) is ArenaBoundary: + continue + if state.get_contact_local_normal(i).y >= FLOOR_NORMAL_MIN_Y: + return true + return false diff --git a/Game/scripts/training_mode.gd b/Game/scripts/training_mode.gd index 0b6dd5a5..8202faca 100644 --- a/Game/scripts/training_mode.gd +++ b/Game/scripts/training_mode.gd @@ -251,8 +251,8 @@ const SHIP_AI_OVERRIDES := [ "ball_touch_reward", "ball_touch_cooldown_ticks", "ball_touch_direction_floor", "velocity_to_ball_weight", "ball_velocity_to_goal_weight", "ball_distance_penalty", "forward_velocity_to_ball_weight", "wall_contact_penalty", "tilt_penalty", - "ground_tilt_penalty", "non_forward_penalty", "speed_reward_weight", "time_penalty", - "airborne_penalty", + "ground_tilt_penalty", "non_forward_penalty", "grounded_upright_reward", + "speed_reward_weight", "time_penalty", "airborne_penalty", ] @@ -313,6 +313,7 @@ func _ai_default(name: String) -> Variant: "tilt_penalty": return 0.0005 "ground_tilt_penalty": return 0.0 "non_forward_penalty": return 0.0 + "grounded_upright_reward": return 0.0 "speed_reward_weight": return 0.004 "time_penalty": return 0.001 "airborne_penalty": return 0.0 diff --git a/training/generation5.py b/training/generation5.py index e9cc3d70..b96736e9 100644 --- a/training/generation5.py +++ b/training/generation5.py @@ -51,7 +51,13 @@ STANDING_ARGS = ["--ent-coef", "0.01", "--entropy-floor"] # full sideways episode now costs ~45, comparable to a goal) and # non_forward_penalty is a new term (ship_ai_controller.gd) directly costing # sideways/reverse planar velocity near the floor, independent of the ball, -# since nothing previously penalized that at all. +# since nothing previously penalized that at all. Both are floor-proximity +# penalties only, with nothing equivalent above GROUND_HANDLING_HEIGHT — on +# its own that risks teaching "avoid the floor" instead of "handle well on +# it", worsening Stage 3's already-airborne-heavy baseline. grounded_upright_ +# reward is the positive counterpart: a bonus for genuine floor contact +# (not just low altitude) while upright, so grounding well is the locally +# profitable choice rather than merely the least-punished one. HANDLING_REWARD_FLAGS = [ "--velocity-to-ball-weight", "0.04", "--forward-velocity-to-ball-weight", "0.06", @@ -63,6 +69,7 @@ HANDLING_REWARD_FLAGS = [ "--tilt-penalty", "0.0002", "--ground-tilt-penalty", "0.05", "--non-forward-penalty", "0.04", + "--grounded-upright-reward", "0.015", ] STAGES = [ diff --git a/training/train.py b/training/train.py index 66d41a07..0682df88 100644 --- a/training/train.py +++ b/training/train.py @@ -363,6 +363,11 @@ def parse_args(): help="Overrides ShipAIController.non_forward_penalty (low-altitude dense cost on sideways/reverse " "planar velocity, independent of the ball)", ) + curriculum.add_argument( + "--grounded-upright-reward", type=float, default=None, + help="Overrides ShipAIController.grounded_upright_reward (dense bonus for genuine floor contact " + "while upright, countering an incentive to just avoid the floor)", + ) curriculum.add_argument( "--speed-reward-weight", type=float, default=None, help="Overrides the orientation-agnostic own-speed reward (generation 5 handling sets it to zero)", @@ -397,6 +402,7 @@ def _curriculum_kwargs(args) -> dict: "ai_tilt_penalty": args.tilt_penalty, "ai_ground_tilt_penalty": args.ground_tilt_penalty, "ai_non_forward_penalty": args.non_forward_penalty, + "ai_grounded_upright_reward": args.grounded_upright_reward, "ai_velocity_to_ball_weight": args.velocity_to_ball_weight, "ai_forward_velocity_to_ball_weight": args.forward_velocity_to_ball_weight, "ai_ball_distance_penalty": args.ball_distance_penalty,