mirror of
https://github.com/jcreek/CosmicClash.git
synced 2026-09-11 00:14:00 +00:00
fca6a46200
Re-ran stage-3 (curric-s3-no_draws vs curric-s2-defend) and the missing stage-4 gate now that the locomotion-mask inference bugfix is in. Both reverse or contradict the pre-fix bookkeeping: curric-s2-defend (grounded) beats curric-s3-no_draws 60-26 and curric-s4-mechanics 57-24 when fairly evaluated, so lifting the locomotion mask in stage 3 was a real regression in floor play, not the improvement the buggy eval reported. Adds a stage-5 "aggression" curriculum entry that resumes from stage 2 directly (via new resume_from_experiment/reference_experiment stage-dict overrides in curriculum.py) instead of compounding the regression through stages 3-4, keeps the locomotion mask on, and retunes ball-pursuit reward weights for much more aggressive floor play. Extends train.py with the three new --velocity-to-ball-weight/--ball-distance-penalty/--ball-touch-reward flags needed to forward that retune to Godot's existing SHIP_AI_OVERRIDES. curriculum_state.json and TRAINING.md are corrected/annotated in place rather than silently rewritten, so the regression stays visible in history.
322 lines
14 KiB
Python
322 lines
14 KiB
Python
"""Orchestrate the staged curriculum (see TRAINING.md's "Curriculum training"
|
|
section): run each stage, evaluate the result against a reference bot, and
|
|
either advance to the next stage or retry the same one.
|
|
|
|
State is persisted to curriculum_state.json (committed to git) so the script
|
|
is safe to Ctrl-C and re-run — it picks up exactly where it left off. Each
|
|
attempt reuses run_training.sh (pull, train, export, commit+push) so every
|
|
attempt's checkpoint, log, and exported policy is versioned like any other
|
|
run; this script additionally evaluates the result and commits the updated
|
|
eval_history.json + curriculum_state.json.
|
|
|
|
The gate is deliberately lenient ("block only on a clear regression," not
|
|
"require improvement") — see TRAINING.md. A 40-episode eval can call a real
|
|
improvement a regression on sample noise alone (this happened with run11:
|
|
it was the first model to deliberately score, but lost its head-to-head
|
|
evals). A strict improvement-required gate would have retried that stage
|
|
forever for the wrong reason. When a stage does fail MAX_RETRIES times in a
|
|
row, the script stops and asks for a human look rather than retrying
|
|
indefinitely or silently advancing past a bad stage.
|
|
|
|
Usage:
|
|
.venv/bin/python curriculum.py # run/resume the curriculum
|
|
.venv/bin/python curriculum.py --seed-checkpoint checkpoints/run11/final.zip
|
|
.venv/bin/python curriculum.py --force-retry # after fixing something, retry the blocked stage
|
|
.venv/bin/python curriculum.py --skip-to-next-stage # human judgment call: good enough, move on anyway
|
|
|
|
Typically started via curriculum.sh, which runs this in a detached tmux
|
|
session the way start_training.sh does for a single run.
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import pathlib
|
|
import subprocess
|
|
import sys
|
|
|
|
TRAINING_DIR = pathlib.Path(__file__).resolve().parent
|
|
STATE_PATH = TRAINING_DIR / "curriculum_state.json"
|
|
EVAL_HISTORY_PATH = TRAINING_DIR / "eval_history.json"
|
|
ROOKIE_REFERENCE = TRAINING_DIR.parent / "Game" / "bots" / "rookie.json"
|
|
|
|
MAX_RETRIES = 2
|
|
EVAL_EPISODES = 100
|
|
# "Clear regression" = the reference beats the candidate by at least this
|
|
# many percentage points of win rate. Below this, noise in a 100-episode
|
|
# sample is a more likely explanation than the stage actually failing (see
|
|
# module docstring) — advance rather than retry.
|
|
REGRESSION_MARGIN = 0.15
|
|
|
|
# Standing flags applied to every attempt, mirroring next_run.sh: reset-std
|
|
# reopens exploration every attempt (harmless on fresh starts — train.py
|
|
# only applies it on --resume), ent-coef keeps it from re-collapsing.
|
|
STANDING_ARGS = ["--reset-std", "0.3", "--ent-coef", "0.001"]
|
|
|
|
STAGES = [
|
|
{
|
|
"name": "score",
|
|
"flags": [
|
|
"--opponent-mode", "inert",
|
|
"--attack-goal-bias", "1.0",
|
|
"--no-allow-vertical", "--no-allow-pitch-roll",
|
|
],
|
|
"grounded": True,
|
|
},
|
|
{
|
|
"name": "defend",
|
|
"flags": [
|
|
"--opponent-mode", "self_play",
|
|
"--no-allow-vertical", "--no-allow-pitch-roll",
|
|
],
|
|
"grounded": True,
|
|
},
|
|
{
|
|
"name": "no_draws",
|
|
"flags": ["--draw-penalty", "5"],
|
|
"grounded": False,
|
|
},
|
|
{
|
|
"name": "mechanics",
|
|
"flags": [],
|
|
"grounded": False,
|
|
},
|
|
{
|
|
"name": "aggression",
|
|
# Deliberately resumes from stage 2 (curric-s2-defend), not stage 4
|
|
# (see resume_from_experiment/reference_experiment below) — the
|
|
# locomotion-mask inference bugfix (8c15c46) revealed that stage 3's
|
|
# full-3D unmask was a clear regression, not an improvement: fairly
|
|
# evaluated, curric-s2-defend beats both curric-s3-no_draws (26-60)
|
|
# and curric-s4-mechanics (24-57). Rather than compound that
|
|
# regression, this stage keeps the locomotion mask ON (matching
|
|
# stage 2's own regime) and just retunes ball-pursuit reward weights,
|
|
# so it can't reopen the same grounded-to-3D transition that caused
|
|
# the earlier failure. Full 3D flight is parked as a separate,
|
|
# later initiative.
|
|
"flags": [
|
|
"--opponent-mode", "self_play",
|
|
"--no-allow-vertical", "--no-allow-pitch-roll",
|
|
"--velocity-to-ball-weight", "0.05", # up from 0.02
|
|
"--ball-distance-penalty", "0.006", # up from 0.002
|
|
"--ball-touch-reward", "0.5", # up from 0.4
|
|
],
|
|
"grounded": True,
|
|
"resume_from_experiment": "curric-s2-defend",
|
|
"reference_experiment": "curric-s2-defend",
|
|
},
|
|
]
|
|
|
|
|
|
def load_state() -> dict:
|
|
if STATE_PATH.exists():
|
|
return json.loads(STATE_PATH.read_text())
|
|
return {"stage_index": 0, "attempt": 0, "status": "in_progress", "log": []}
|
|
|
|
|
|
def save_state(state: dict) -> None:
|
|
STATE_PATH.write_text(json.dumps(state, indent=2) + "\n")
|
|
|
|
|
|
def experiment_name(stage_index: int, attempt: int) -> str:
|
|
name = f"curric-s{stage_index + 1}-{STAGES[stage_index]['name']}"
|
|
return name if attempt == 0 else f"{name}-retry{attempt}"
|
|
|
|
|
|
def resume_checkpoint(stage_index: int, attempt: int, seed_checkpoint: str | None) -> str | None:
|
|
if attempt > 0:
|
|
# Retry: keep training the same stage's own last attempt.
|
|
prev = experiment_name(stage_index, attempt - 1)
|
|
return str(TRAINING_DIR / "checkpoints" / prev / "final.zip")
|
|
if stage_index == 0:
|
|
# Deliberately fresh by default: the curriculum exists because
|
|
# resuming self-play across a regime change (run10, run11) didn't
|
|
# work, so stage 1 starts from a random policy under its own
|
|
# regime unless --seed-checkpoint says otherwise.
|
|
return seed_checkpoint
|
|
prev_experiment = _resume_source_experiment(stage_index)
|
|
return str(TRAINING_DIR / "checkpoints" / prev_experiment / "final.zip")
|
|
|
|
|
|
def reference_bot(stage_index: int) -> str:
|
|
if stage_index == 0:
|
|
return str(ROOKIE_REFERENCE)
|
|
prev_experiment = _reference_source_experiment(stage_index)
|
|
return str(TRAINING_DIR.parent / "Game" / "bots" / f"{prev_experiment}.json")
|
|
|
|
|
|
# A stage normally chains off "whatever passed at the previous index," but a
|
|
# stage can instead name an explicit resume_from_experiment/reference_experiment
|
|
# to skip a since-regressed branch (see the "aggression" stage) without
|
|
# rewriting history for the stages it's skipping past.
|
|
def _resume_source_experiment(stage_index: int) -> str:
|
|
override = STAGES[stage_index].get("resume_from_experiment")
|
|
return override if override else _passing_experiment_for_stage(stage_index - 1)
|
|
|
|
|
|
def _reference_source_experiment(stage_index: int) -> str:
|
|
override = STAGES[stage_index].get("reference_experiment")
|
|
return override if override else _passing_experiment_for_stage(stage_index - 1)
|
|
|
|
|
|
def _passing_experiment_for_stage(stage_index: int) -> str:
|
|
state = load_state()
|
|
for entry in state["log"]:
|
|
if entry["stage_index"] == stage_index and entry["decision"] == "pass":
|
|
return entry["experiment"]
|
|
raise RuntimeError(f"No passing attempt recorded for stage {stage_index} ({STAGES[stage_index]['name']})")
|
|
|
|
|
|
def _grounded_for_experiment(experiment: str) -> bool:
|
|
if experiment == "rookie":
|
|
return False
|
|
for index, stage in enumerate(STAGES):
|
|
if experiment_name(index, 0) == experiment:
|
|
return stage["grounded"]
|
|
raise ValueError(f"Unknown experiment for groundedness lookup: {experiment}")
|
|
|
|
|
|
def run_stage_attempt(stage_index: int, attempt: int, args) -> str:
|
|
exp = experiment_name(stage_index, attempt)
|
|
resume = resume_checkpoint(stage_index, attempt, args.seed_checkpoint)
|
|
cmd = [
|
|
"./run_training.sh", exp,
|
|
"--timesteps", str(args.timesteps),
|
|
"--n-parallel", str(args.n_parallel),
|
|
"--speedup", str(args.speedup),
|
|
*STANDING_ARGS,
|
|
*STAGES[stage_index]["flags"],
|
|
]
|
|
if resume:
|
|
cmd += ["--resume", resume]
|
|
print(f"\n=== Stage {stage_index + 1}/{len(STAGES)} ({STAGES[stage_index]['name']}), "
|
|
f"attempt {attempt + 1}/{MAX_RETRIES + 1}: {exp} ===")
|
|
print(" ".join(cmd))
|
|
subprocess.run(cmd, cwd=TRAINING_DIR, check=True)
|
|
return exp
|
|
|
|
|
|
def reference_grounded(stage_index: int) -> bool:
|
|
if stage_index == 0:
|
|
# rookie.json predates the locomotion mask entirely — always full 3D.
|
|
return False
|
|
return _grounded_for_experiment(_reference_source_experiment(stage_index))
|
|
|
|
|
|
def evaluate_attempt(experiment: str, reference: str, episodes: int, stage_index: int) -> dict:
|
|
candidate = TRAINING_DIR.parent / "Game" / "bots" / f"{experiment}.json"
|
|
cmd = [".venv/bin/python", "evaluate.py", str(candidate), reference, "--episodes", str(episodes)]
|
|
# Must match how each side was actually trained — see ai_ship_controller.gd's
|
|
# allow_vertical/allow_pitch_roll and evaluate.py's --grounded-a/-b.
|
|
if STAGES[stage_index]["grounded"]:
|
|
cmd.append("--grounded-a")
|
|
if reference_grounded(stage_index):
|
|
cmd.append("--grounded-b")
|
|
print(" ".join(cmd))
|
|
subprocess.run(cmd, cwd=TRAINING_DIR, check=True)
|
|
history = json.loads(EVAL_HISTORY_PATH.read_text())
|
|
return history[-1]
|
|
|
|
|
|
def decide(record: dict) -> str:
|
|
win_rate_candidate = record["wins_a"] / record["episodes"]
|
|
win_rate_reference = record["wins_b"] / record["episodes"]
|
|
if win_rate_reference - win_rate_candidate >= REGRESSION_MARGIN:
|
|
return "fail"
|
|
return "pass"
|
|
|
|
|
|
def commit_progress(experiment: str) -> None:
|
|
subprocess.run(["git", "add", "curriculum_state.json", "eval_history.json"], cwd=TRAINING_DIR, check=True)
|
|
result = subprocess.run(["git", "diff", "--cached", "--quiet"], cwd=TRAINING_DIR)
|
|
if result.returncode == 0:
|
|
return
|
|
subprocess.run(
|
|
["git", "commit", "-m", f"chore(training): curriculum progress after {experiment}"],
|
|
cwd=TRAINING_DIR, check=True,
|
|
)
|
|
subprocess.run(["git", "push"], cwd=TRAINING_DIR, check=True)
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
parser.add_argument("--timesteps", type=int, default=20_000_000)
|
|
parser.add_argument("--n-parallel", type=int, default=14)
|
|
parser.add_argument("--speedup", type=int, default=16)
|
|
parser.add_argument("--seed-checkpoint", default=None, help="Resume stage 1 from this checkpoint instead of a fresh policy")
|
|
parser.add_argument("--force-retry", action="store_true", help="Retry a blocked stage after human review")
|
|
parser.add_argument("--skip-to-next-stage", action="store_true", help="Human judgment call: treat the blocked stage as good enough, advance anyway")
|
|
args = parser.parse_args()
|
|
|
|
state = load_state()
|
|
|
|
if state["status"] == "blocked":
|
|
if args.skip_to_next_stage:
|
|
print(f"Human override: advancing past stage {state['stage_index'] + 1} "
|
|
f"({STAGES[state['stage_index']]['name']}) despite exhausted retries.")
|
|
state["log"].append({
|
|
"stage_index": state["stage_index"], "experiment": experiment_name(state["stage_index"], state["attempt"]),
|
|
"attempt": state["attempt"], "decision": "pass", "override": "skip_to_next_stage",
|
|
})
|
|
state["stage_index"] += 1
|
|
state["attempt"] = 0
|
|
state["status"] = "in_progress"
|
|
save_state(state)
|
|
elif args.force_retry:
|
|
print(f"Human override: retrying stage {state['stage_index'] + 1} "
|
|
f"({STAGES[state['stage_index']]['name']}) after review.")
|
|
state["attempt"] += 1
|
|
state["status"] = "in_progress"
|
|
save_state(state)
|
|
else:
|
|
print(f"BLOCKED at stage {state['stage_index'] + 1} ({STAGES[state['stage_index']]['name']}) "
|
|
f"after {MAX_RETRIES + 1} attempts — see curriculum_state.json's log for eval results.")
|
|
print("Re-run with --force-retry (after adjusting flags/timesteps) or "
|
|
"--skip-to-next-stage (advance anyway) once you've looked at why.")
|
|
sys.exit(1)
|
|
|
|
while state["stage_index"] < len(STAGES):
|
|
stage_index = state["stage_index"]
|
|
attempt = state["attempt"]
|
|
|
|
experiment = run_stage_attempt(stage_index, attempt, args)
|
|
reference = reference_bot(stage_index)
|
|
record = evaluate_attempt(experiment, reference, EVAL_EPISODES, stage_index)
|
|
decision = decide(record)
|
|
|
|
print(f"{experiment}: candidate {record['wins_a']}-{record['wins_b']} reference "
|
|
f"({record['draws']} draws) over {record['episodes']} episodes -> {decision}")
|
|
|
|
state["log"].append({
|
|
"stage_index": stage_index, "experiment": experiment, "attempt": attempt,
|
|
"eval": record, "decision": decision,
|
|
})
|
|
|
|
if decision == "pass":
|
|
state["stage_index"] += 1
|
|
state["attempt"] = 0
|
|
state["status"] = "in_progress"
|
|
save_state(state)
|
|
commit_progress(experiment)
|
|
continue
|
|
|
|
if attempt >= MAX_RETRIES:
|
|
state["status"] = "blocked"
|
|
save_state(state)
|
|
commit_progress(experiment)
|
|
print(f"\nBLOCKED: stage {stage_index + 1} ({STAGES[stage_index]['name']}) failed "
|
|
f"{MAX_RETRIES + 1} attempts in a row. Stopping for human review — see "
|
|
f"curriculum_state.json. Re-run with --force-retry or --skip-to-next-stage.")
|
|
sys.exit(1)
|
|
|
|
state["attempt"] += 1
|
|
save_state(state)
|
|
commit_progress(experiment)
|
|
|
|
print("\nCurriculum complete — all stages passed.")
|
|
state["status"] = "done"
|
|
save_state(state)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|