feat(*): Log live goal rate to TensorBoard during training

This commit is contained in:
Josh Creek
2026-07-28 21:33:41 +01:00
parent c998d2271a
commit bca08d266e
4 changed files with 50 additions and 4 deletions
+17
View File
@@ -107,6 +107,15 @@ var ball: RigidBody3D
var opponent: Ship
var attack_goal_position: Vector3
# Set directly by TrainingMode (_on_goal_scored / the timeout branch in
# _physics_process) at the same time as `done = true`. Deliberately NOT
# cleared in reset(): TrainingMode's _reset_episode() (which calls reset())
# runs synchronously, immediately after done is set, before the Sync node
# ever reads get_info()/get_done() for that terminal tick — clearing it here
# would wipe the value that read needs. Both call sites always overwrite
# (true on goal, false on timeout) rather than toggle, so no reset is needed.
var goal_scored_this_episode := false
var _ticks_since_ball_touch := 1 << 30 # large so the first touch always pays
@@ -135,6 +144,14 @@ func get_reward() -> float:
return reward
# Symmetric across both self-play agents: reports whether this episode ended
# in a goal at all, not which team scored — a clean "goal rate" signal
# distinct from rollout/ep_rew_mean, which mixes this with dense shaping
# (ball chasing/touching). See train.py's GoalRateCallback.
func get_info() -> Dictionary:
return {"goal_scored": goal_scored_this_episode}
func get_action_space() -> Dictionary:
return {
"thrust": {"size": 3, "action_type": "continuous"},
+2
View File
@@ -290,6 +290,7 @@ func _physics_process(_delta):
for agent in _agents:
agent.reward -= draw_penalty
agent.done = true
agent.goal_scored_this_episode = false
_reset_episode()
return
@@ -320,6 +321,7 @@ func _on_goal_scored(conceding_team: int) -> void:
for agent in _agents:
agent.reward += goal_reward if agent.ship.team != conceding_team else -goal_reward
agent.done = true
agent.goal_scored_this_episode = true
_reset_episode()