mirror of
https://github.com/jcreek/CosmicClash.git
synced 2026-09-14 13:42:09 +00:00
fix(training): correct stage-3 eval (locomotion-mask bugfix) and add grounded aggression stage
Re-ran stage-3 (curric-s3-no_draws vs curric-s2-defend) and the missing stage-4 gate now that the locomotion-mask inference bugfix is in. Both reverse or contradict the pre-fix bookkeeping: curric-s2-defend (grounded) beats curric-s3-no_draws 60-26 and curric-s4-mechanics 57-24 when fairly evaluated, so lifting the locomotion mask in stage 3 was a real regression in floor play, not the improvement the buggy eval reported. Adds a stage-5 "aggression" curriculum entry that resumes from stage 2 directly (via new resume_from_experiment/reference_experiment stage-dict overrides in curriculum.py) instead of compounding the regression through stages 3-4, keeps the locomotion mask on, and retunes ball-pursuit reward weights for much more aggressive floor play. Extends train.py with the three new --velocity-to-ball-weight/--ball-distance-penalty/--ball-touch-reward flags needed to forward that retune to Godot's existing SHIP_AI_OVERRIDES. curriculum_state.json and TRAINING.md are corrected/annotated in place rather than silently rewritten, so the regression stays visible in history.
This commit is contained in:
+53
-5
@@ -80,6 +80,30 @@ STAGES = [
|
||||
"flags": [],
|
||||
"grounded": False,
|
||||
},
|
||||
{
|
||||
"name": "aggression",
|
||||
# Deliberately resumes from stage 2 (curric-s2-defend), not stage 4
|
||||
# (see resume_from_experiment/reference_experiment below) — the
|
||||
# locomotion-mask inference bugfix (8c15c46) revealed that stage 3's
|
||||
# full-3D unmask was a clear regression, not an improvement: fairly
|
||||
# evaluated, curric-s2-defend beats both curric-s3-no_draws (26-60)
|
||||
# and curric-s4-mechanics (24-57). Rather than compound that
|
||||
# regression, this stage keeps the locomotion mask ON (matching
|
||||
# stage 2's own regime) and just retunes ball-pursuit reward weights,
|
||||
# so it can't reopen the same grounded-to-3D transition that caused
|
||||
# the earlier failure. Full 3D flight is parked as a separate,
|
||||
# later initiative.
|
||||
"flags": [
|
||||
"--opponent-mode", "self_play",
|
||||
"--no-allow-vertical", "--no-allow-pitch-roll",
|
||||
"--velocity-to-ball-weight", "0.05", # up from 0.02
|
||||
"--ball-distance-penalty", "0.006", # up from 0.002
|
||||
"--ball-touch-reward", "0.5", # up from 0.4
|
||||
],
|
||||
"grounded": True,
|
||||
"resume_from_experiment": "curric-s2-defend",
|
||||
"reference_experiment": "curric-s2-defend",
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
@@ -109,18 +133,31 @@ def resume_checkpoint(stage_index: int, attempt: int, seed_checkpoint: str | Non
|
||||
# work, so stage 1 starts from a random policy under its own
|
||||
# regime unless --seed-checkpoint says otherwise.
|
||||
return seed_checkpoint
|
||||
prev_stage = STAGES[stage_index - 1]["name"]
|
||||
prev_experiment = _passing_experiment_for_stage(stage_index - 1)
|
||||
prev_experiment = _resume_source_experiment(stage_index)
|
||||
return str(TRAINING_DIR / "checkpoints" / prev_experiment / "final.zip")
|
||||
|
||||
|
||||
def reference_bot(stage_index: int) -> str:
|
||||
if stage_index == 0:
|
||||
return str(ROOKIE_REFERENCE)
|
||||
prev_experiment = _passing_experiment_for_stage(stage_index - 1)
|
||||
prev_experiment = _reference_source_experiment(stage_index)
|
||||
return str(TRAINING_DIR.parent / "Game" / "bots" / f"{prev_experiment}.json")
|
||||
|
||||
|
||||
# A stage normally chains off "whatever passed at the previous index," but a
|
||||
# stage can instead name an explicit resume_from_experiment/reference_experiment
|
||||
# to skip a since-regressed branch (see the "aggression" stage) without
|
||||
# rewriting history for the stages it's skipping past.
|
||||
def _resume_source_experiment(stage_index: int) -> str:
|
||||
override = STAGES[stage_index].get("resume_from_experiment")
|
||||
return override if override else _passing_experiment_for_stage(stage_index - 1)
|
||||
|
||||
|
||||
def _reference_source_experiment(stage_index: int) -> str:
|
||||
override = STAGES[stage_index].get("reference_experiment")
|
||||
return override if override else _passing_experiment_for_stage(stage_index - 1)
|
||||
|
||||
|
||||
def _passing_experiment_for_stage(stage_index: int) -> str:
|
||||
state = load_state()
|
||||
for entry in state["log"]:
|
||||
@@ -129,6 +166,15 @@ def _passing_experiment_for_stage(stage_index: int) -> str:
|
||||
raise RuntimeError(f"No passing attempt recorded for stage {stage_index} ({STAGES[stage_index]['name']})")
|
||||
|
||||
|
||||
def _grounded_for_experiment(experiment: str) -> bool:
|
||||
if experiment == "rookie":
|
||||
return False
|
||||
for index, stage in enumerate(STAGES):
|
||||
if experiment_name(index, 0) == experiment:
|
||||
return stage["grounded"]
|
||||
raise ValueError(f"Unknown experiment for groundedness lookup: {experiment}")
|
||||
|
||||
|
||||
def run_stage_attempt(stage_index: int, attempt: int, args) -> str:
|
||||
exp = experiment_name(stage_index, attempt)
|
||||
resume = resume_checkpoint(stage_index, attempt, args.seed_checkpoint)
|
||||
@@ -150,8 +196,10 @@ def run_stage_attempt(stage_index: int, attempt: int, args) -> str:
|
||||
|
||||
|
||||
def reference_grounded(stage_index: int) -> bool:
|
||||
# rookie.json predates the locomotion mask entirely — always full 3D.
|
||||
return False if stage_index == 0 else STAGES[stage_index - 1]["grounded"]
|
||||
if stage_index == 0:
|
||||
# rookie.json predates the locomotion mask entirely — always full 3D.
|
||||
return False
|
||||
return _grounded_for_experiment(_reference_source_experiment(stage_index))
|
||||
|
||||
|
||||
def evaluate_attempt(experiment: str, reference: str, episodes: int, stage_index: int) -> dict:
|
||||
|
||||
Reference in New Issue
Block a user