mirror of
https://github.com/jcreek/CosmicClash.git
synced 2026-09-11 08:23:45 +00:00
efda6c1a05
Generation 2's first two real stage-1 attempts both independently restarted from curric-s5-aggression (reset_retry_checkpoint) with identical flags and landed at 32% and 27% win rate vs the reference -- a real regression either way, but too much spread between "identical" runs for repeat fresh restarts to be a controlled test of anything. The first attempt's own trajectory (ep_rew_mean climbing from -10.86 toward ~0 by the 240M-step cutoff, briefly touching positive) looked closer to convergence than the second's, so retries now continue that attempt's own checkpoint for another full timesteps budget instead of resetting to foundation again. Drops retry1 and retry2 (checkpoints, logs, exported bots, eval_history entries) -- retry2 never trained meaningfully before crashing on the GoalRateCallback bug just fixed, and retry1 was the inferior of the two real samples. curriculum_state.json rewinds to attempt 1, in_progress, so the next run resumes 20260726-1904-curric-s1-unmask/final.zip directly.
413 lines
20 KiB
Python
413 lines
20 KiB
Python
"""Orchestrate the staged curriculum (see TRAINING.md's "Curriculum training"
|
|
section): run each stage, evaluate the result against a reference bot, and
|
|
either advance to the next stage or retry the same one.
|
|
|
|
State is persisted to curriculum_state.json (committed to git) so the script
|
|
is safe to Ctrl-C and re-run — it picks up exactly where it left off. Each
|
|
attempt reuses run_training.sh (pull, train, export, commit+push) so every
|
|
attempt's checkpoint, log, and exported policy is versioned like any other
|
|
run; this script additionally evaluates the result and commits the updated
|
|
eval_history.json + curriculum_state.json.
|
|
|
|
The gate is deliberately lenient ("block only on a clear regression," not
|
|
"require improvement") — see TRAINING.md. A 40-episode eval can call a real
|
|
improvement a regression on sample noise alone (this happened with run11:
|
|
it was the first model to deliberately score, but lost its head-to-head
|
|
evals). A strict improvement-required gate would have retried that stage
|
|
forever for the wrong reason. When a stage does fail MAX_RETRIES times in a
|
|
row, the script stops and asks for a human look rather than retrying
|
|
indefinitely or silently advancing past a bad stage.
|
|
|
|
This is generation 2 of the curriculum. Generation 1 (6 stages: score,
|
|
defend, no_draws, mechanics, aggression, unmask) ran 2026-07-21 through
|
|
2026-07-26 and is archived in curriculum_state_gen1.json — its final stage
|
|
("unmask", full 3D flight on top of the aggression retune) failed 3 straight
|
|
attempts, monotonically worsening (25% -> 20% -> 15% win rate vs
|
|
curric-s5-aggression) because every retry resumed the same drifting
|
|
checkpoint under identical flags instead of actually changing anything.
|
|
Rather than let generation 1's stage numbering grow indefinitely
|
|
(unmask-retry4, retry5, ...), generation 2 starts a fresh stage 1 seeded
|
|
directly from curric-s5-aggression's own checkpoint (FOUNDATION_EXPERIMENT
|
|
below) — the last stage that actually passed cleanly — carrying over its
|
|
trained progress without re-running stages 1-5. See TRAINING.md for the
|
|
full generation 1 history and generation 2's design.
|
|
|
|
Every experiment name this script generates is timestamped
|
|
(YYYYMMDD-HHMM-<name>, applied once in run_stage_attempt) so runs stay
|
|
unique across restarts/generations and sort chronologically in TensorBoard
|
|
and checkpoints/ — plain names like "curric-s1-score" from generation 1
|
|
would otherwise collide with generation 2's own stage 1.
|
|
|
|
Usage:
|
|
.venv/bin/python curriculum.py # run/resume the curriculum
|
|
.venv/bin/python curriculum.py --seed-checkpoint checkpoints/run11/final.zip
|
|
.venv/bin/python curriculum.py --force-retry # after fixing something, retry the blocked stage
|
|
.venv/bin/python curriculum.py --skip-to-next-stage # human judgment call: good enough, move on anyway
|
|
|
|
Typically started via curriculum.sh, which runs this in a detached tmux
|
|
session the way start_training.sh does for a single run.
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import pathlib
|
|
import subprocess
|
|
import sys
|
|
from datetime import datetime
|
|
|
|
TRAINING_DIR = pathlib.Path(__file__).resolve().parent
|
|
STATE_PATH = TRAINING_DIR / "curriculum_state.json"
|
|
EVAL_HISTORY_PATH = TRAINING_DIR / "eval_history.json"
|
|
ROOKIE_REFERENCE = TRAINING_DIR.parent / "Game" / "bots" / "rookie.json"
|
|
|
|
# Generation 1's last cleanly-passing checkpoint (see curriculum_state_gen1.json)
|
|
# — generation 2's stage 1 builds on this directly instead of re-running
|
|
# stages 1-5.
|
|
FOUNDATION_EXPERIMENT = "curric-s5-aggression"
|
|
|
|
# Groundedness (locomotion-mask state) for experiments that predate this
|
|
# generation's own log, so _grounded_for_experiment can still answer for
|
|
# them — see that function.
|
|
LEGACY_GROUNDED = {
|
|
"rookie": False,
|
|
FOUNDATION_EXPERIMENT: True,
|
|
}
|
|
|
|
MAX_RETRIES = 2
|
|
EVAL_EPISODES = 100
|
|
# "Clear regression" = the reference beats the candidate by at least this
|
|
# many percentage points of win rate. Below this, noise in a 100-episode
|
|
# sample is a more likely explanation than the stage actually failing (see
|
|
# module docstring) — advance rather than retry.
|
|
REGRESSION_MARGIN = 0.15
|
|
|
|
# Standing flags applied to every attempt, mirroring next_run.sh: reset-std
|
|
# reopens exploration every attempt (harmless on fresh starts — train.py
|
|
# only applies it on --resume), ent-coef keeps it from re-collapsing.
|
|
STANDING_ARGS = ["--reset-std", "0.3", "--ent-coef", "0.001"]
|
|
|
|
STAGES = [
|
|
{
|
|
"name": "unmask",
|
|
# Re-opens full 3D controls (no more --no-allow-vertical/
|
|
# --no-allow-pitch-roll) on top of the aggression retune, instead of
|
|
# keeping locomotion masked indefinitely. The mask blocked *thrust*-
|
|
# driven flight outright; airborne_penalty (dense, scaled by height
|
|
# above the floor — see ship_ai_controller.gd) is meant to teach the
|
|
# policy to prefer staying grounded through incentives rather than a
|
|
# hard constraint, so it can start learning when the other axes are
|
|
# actually useful (aerial saves, wall recoveries) instead of never
|
|
# touching them.
|
|
#
|
|
# Generation 1 ran this exact transition 3 times (unmask, retry1,
|
|
# retry2) with identical flags and got monotonically worse each time
|
|
# (25% -> 20% -> 15% win rate vs curric-s5-aggression) — a blind
|
|
# retry just continues training the same drifting policy for
|
|
# longer, it was never going to converge differently. An adversarial
|
|
# review of a first patch (two modest new flags, still resuming the
|
|
# drifted retry2 checkpoint) found that insufficient too: the resume
|
|
# target was the worst of the three already-degraded checkpoints,
|
|
# and the new weights were too small to compete with the unchanged
|
|
# ball-pursuit terms. Generation 2's stage 1 instead:
|
|
# - resumes from FOUNDATION_EXPERIMENT (curric-s5-aggression)
|
|
# directly (resume_from_experiment below) for the first attempt.
|
|
# - raises velocity_to_ball_weight and ball_distance_penalty
|
|
# further (the actual ball-chasing terms, unchanged since stage
|
|
# 5 despite three failed attempts) and ball_touch_reward
|
|
# alongside them.
|
|
# - raises ball_velocity_to_goal_weight (reward for moving the
|
|
# ball toward the goal, not just touching it) and goal_reward
|
|
# (the terminal reward for scoring) — both newly exposed via
|
|
# train.py, previously only reachable as raw Godot cmdline args.
|
|
# - adds draw_penalty (proven effective in generation 1's stage 3
|
|
# against passivity), which this transition had never set:
|
|
# previously all carrot for scoring, no stick for never scoring.
|
|
"flags": [
|
|
"--opponent-mode", "self_play",
|
|
"--velocity-to-ball-weight", "0.08", # up from 0.05
|
|
"--ball-distance-penalty", "0.01", # up from 0.006
|
|
"--ball-touch-reward", "0.7", # up from 0.5
|
|
"--airborne-penalty", "0.003",
|
|
"--ball-velocity-to-goal-weight", "0.06", # up from 0.02 (0.004 default)
|
|
"--goal-reward", "80", # up from 60 (40 default)
|
|
"--draw-penalty", "5",
|
|
],
|
|
# 2026-07-29: generation 2's own first two attempts (both independently
|
|
# resumed from FOUNDATION_EXPERIMENT under reset_retry_checkpoint,
|
|
# identical flags/budget) landed at 32% and 27% win rate — a real
|
|
# regression either way, but with enough run-to-run spread that
|
|
# "identical fresh restart" isn't a controlled test of anything. The
|
|
# first attempt's own trajectory (ep_rew_mean climbing from -10.86
|
|
# toward ~0 by the 240M-step cutoff, briefly touching positive) looked
|
|
# closer to convergence than the second's, so rather than another
|
|
# independent restart from foundation, retries now continue *that*
|
|
# attempt's own checkpoint for another full timesteps budget — an
|
|
# actual test of "did it just need more time," not another coin flip.
|
|
# A third, unrelated attempt crashed immediately (see train.py's
|
|
# GoalRateCallback KeyError fix) before contributing any real signal
|
|
# and was discarded rather than counted.
|
|
"grounded": False,
|
|
"timesteps": 240_000_000, # ~24h at the standing n-parallel/speedup (20M took ~2h)
|
|
"resume_from_experiment": FOUNDATION_EXPERIMENT,
|
|
"reference_experiment": FOUNDATION_EXPERIMENT,
|
|
},
|
|
]
|
|
|
|
|
|
def load_state() -> dict:
|
|
if STATE_PATH.exists():
|
|
return json.loads(STATE_PATH.read_text())
|
|
return {"stage_index": 0, "attempt": 0, "status": "in_progress", "log": []}
|
|
|
|
|
|
def save_state(state: dict) -> None:
|
|
STATE_PATH.write_text(json.dumps(state, indent=2) + "\n")
|
|
|
|
|
|
def experiment_name(stage_index: int, attempt: int) -> str:
|
|
name = f"curric-s{stage_index + 1}-{STAGES[stage_index]['name']}"
|
|
return name if attempt == 0 else f"{name}-retry{attempt}"
|
|
|
|
|
|
def _logged_experiment_name(stage_index: int, attempt: int) -> str:
|
|
"""The actual (timestamped) experiment name recorded when this attempt
|
|
ran — needed anywhere a *past* attempt's real name matters, since
|
|
experiment_name() alone no longer identifies a run on disk (see
|
|
run_stage_attempt's timestamp prefix)."""
|
|
state = load_state()
|
|
for entry in state["log"]:
|
|
if entry["stage_index"] == stage_index and entry["attempt"] == attempt:
|
|
return entry["experiment"]
|
|
raise RuntimeError(f"No logged experiment for stage {stage_index} attempt {attempt}")
|
|
|
|
|
|
def resume_checkpoint(stage_index: int, attempt: int, seed_checkpoint: str | None) -> str | None:
|
|
if attempt > 0 and not STAGES[stage_index].get("reset_retry_checkpoint"):
|
|
# Retry: keep training the same stage's own last attempt.
|
|
prev = _logged_experiment_name(stage_index, attempt - 1)
|
|
return str(TRAINING_DIR / "checkpoints" / prev / "final.zip")
|
|
if stage_index == 0 and seed_checkpoint:
|
|
return seed_checkpoint
|
|
if stage_index == 0 and not STAGES[0].get("resume_from_experiment"):
|
|
# Deliberately fresh by default: the curriculum exists because
|
|
# resuming self-play across a regime change (run10, run11) didn't
|
|
# work, so a from-scratch stage 1 starts from a random policy under
|
|
# its own regime unless --seed-checkpoint or resume_from_experiment
|
|
# says otherwise.
|
|
return None
|
|
# Either a later stage chaining off its predecessor, or
|
|
# reset_retry_checkpoint: this stage's own retries have been drifting
|
|
# rather than converging (see the "unmask" stage's comment) — resume
|
|
# from the stage's normal resume source instead of compounding the last
|
|
# failed attempt's drift.
|
|
prev_experiment = _resume_source_experiment(stage_index)
|
|
return str(TRAINING_DIR / "checkpoints" / prev_experiment / "final.zip")
|
|
|
|
|
|
def reference_bot(stage_index: int) -> str:
|
|
if stage_index == 0 and not STAGES[0].get("reference_experiment"):
|
|
return str(ROOKIE_REFERENCE)
|
|
prev_experiment = _reference_source_experiment(stage_index)
|
|
return str(TRAINING_DIR.parent / "Game" / "bots" / f"{prev_experiment}.json")
|
|
|
|
|
|
# A stage normally chains off "whatever passed at the previous index," but a
|
|
# stage can instead name an explicit resume_from_experiment/reference_experiment
|
|
# to skip a since-regressed branch, or (stage 0) to seed from a fixed
|
|
# foundation checkpoint instead of a from-scratch policy.
|
|
def _resume_source_experiment(stage_index: int) -> str:
|
|
override = STAGES[stage_index].get("resume_from_experiment")
|
|
return override if override else _passing_experiment_for_stage(stage_index - 1)
|
|
|
|
|
|
def _reference_source_experiment(stage_index: int) -> str:
|
|
override = STAGES[stage_index].get("reference_experiment")
|
|
return override if override else _passing_experiment_for_stage(stage_index - 1)
|
|
|
|
|
|
def _passing_experiment_for_stage(stage_index: int) -> str:
|
|
state = load_state()
|
|
for entry in state["log"]:
|
|
if entry["stage_index"] == stage_index and entry["decision"] == "pass":
|
|
return entry["experiment"]
|
|
raise RuntimeError(f"No passing attempt recorded for stage {stage_index} ({STAGES[stage_index]['name']})")
|
|
|
|
|
|
def _grounded_for_experiment(experiment: str) -> bool:
|
|
if experiment in LEGACY_GROUNDED:
|
|
return LEGACY_GROUNDED[experiment]
|
|
state = load_state()
|
|
for entry in state["log"]:
|
|
if entry["experiment"] == experiment:
|
|
return STAGES[entry["stage_index"]]["grounded"]
|
|
raise ValueError(f"Unknown experiment for groundedness lookup: {experiment}")
|
|
|
|
|
|
def run_stage_attempt(stage_index: int, attempt: int, args) -> str:
|
|
# Timestamped so names stay unique across restarts/generations and sort
|
|
# chronologically in TensorBoard/checkpoints — see module docstring.
|
|
exp = f"{datetime.now().strftime('%Y%m%d-%H%M')}-{experiment_name(stage_index, attempt)}"
|
|
resume = resume_checkpoint(stage_index, attempt, args.seed_checkpoint)
|
|
# A stage can override the run's timesteps budget (see "floor-lock",
|
|
# which deliberately runs much longer than the ~20M/~2h every stage so
|
|
# far has used); otherwise it falls back to curriculum.py's own --timesteps.
|
|
timesteps = STAGES[stage_index].get("timesteps", args.timesteps)
|
|
cmd = [
|
|
"./run_training.sh", exp,
|
|
"--timesteps", str(timesteps),
|
|
"--n-parallel", str(args.n_parallel),
|
|
"--speedup", str(args.speedup),
|
|
*STANDING_ARGS,
|
|
*STAGES[stage_index]["flags"],
|
|
]
|
|
if resume:
|
|
cmd += ["--resume", resume]
|
|
print(f"\n=== Stage {stage_index + 1}/{len(STAGES)} ({STAGES[stage_index]['name']}), "
|
|
f"attempt {attempt + 1}/{MAX_RETRIES + 1}: {exp} ===")
|
|
print(" ".join(cmd))
|
|
subprocess.run(cmd, cwd=TRAINING_DIR, check=True)
|
|
return exp
|
|
|
|
|
|
def reference_grounded(stage_index: int) -> bool:
|
|
if stage_index == 0 and not STAGES[0].get("reference_experiment"):
|
|
# rookie.json predates the locomotion mask entirely — always full 3D.
|
|
return False
|
|
return _grounded_for_experiment(_reference_source_experiment(stage_index))
|
|
|
|
|
|
def evaluate_attempt(experiment: str, reference: str, episodes: int, stage_index: int) -> dict:
|
|
candidate = TRAINING_DIR.parent / "Game" / "bots" / f"{experiment}.json"
|
|
cmd = [".venv/bin/python", "evaluate.py", str(candidate), reference, "--episodes", str(episodes)]
|
|
# Must match how each side was actually trained — see ai_ship_controller.gd's
|
|
# allow_vertical/allow_pitch_roll and evaluate.py's --grounded-a/-b.
|
|
if STAGES[stage_index]["grounded"]:
|
|
cmd.append("--grounded-a")
|
|
if reference_grounded(stage_index):
|
|
cmd.append("--grounded-b")
|
|
print(" ".join(cmd))
|
|
subprocess.run(cmd, cwd=TRAINING_DIR, check=True)
|
|
history = json.loads(EVAL_HISTORY_PATH.read_text())
|
|
return history[-1]
|
|
|
|
|
|
def decide(record: dict) -> str:
|
|
win_rate_candidate = record["wins_a"] / record["episodes"]
|
|
win_rate_reference = record["wins_b"] / record["episodes"]
|
|
if win_rate_reference - win_rate_candidate >= REGRESSION_MARGIN:
|
|
return "fail"
|
|
return "pass"
|
|
|
|
|
|
def commit_progress(experiment: str) -> None:
|
|
subprocess.run(["git", "add", "curriculum_state.json", "eval_history.json"], cwd=TRAINING_DIR, check=True)
|
|
result = subprocess.run(["git", "diff", "--cached", "--quiet"], cwd=TRAINING_DIR)
|
|
if result.returncode == 0:
|
|
return
|
|
subprocess.run(
|
|
["git", "commit", "-m", f"chore(training): curriculum progress after {experiment}"],
|
|
cwd=TRAINING_DIR, check=True,
|
|
)
|
|
subprocess.run(["git", "push"], cwd=TRAINING_DIR, check=True)
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
parser.add_argument("--timesteps", type=int, default=20_000_000)
|
|
parser.add_argument("--n-parallel", type=int, default=14)
|
|
parser.add_argument("--speedup", type=int, default=16)
|
|
parser.add_argument(
|
|
"--seed-checkpoint", default=None,
|
|
help="Resume stage 1 from this checkpoint instead of its default resume source "
|
|
"(FOUNDATION_EXPERIMENT's checkpoint)",
|
|
)
|
|
parser.add_argument("--force-retry", action="store_true", help="Retry a blocked stage after human review")
|
|
parser.add_argument("--skip-to-next-stage", action="store_true", help="Human judgment call: treat the blocked stage as good enough, advance anyway")
|
|
args = parser.parse_args()
|
|
|
|
state = load_state()
|
|
|
|
if state["status"] == "blocked":
|
|
if args.skip_to_next_stage:
|
|
print(f"Human override: advancing past stage {state['stage_index'] + 1} "
|
|
f"({STAGES[state['stage_index']]['name']}) despite exhausted retries.")
|
|
skipped_experiment = _logged_experiment_name(state["stage_index"], state["attempt"])
|
|
state["log"].append({
|
|
"stage_index": state["stage_index"],
|
|
"experiment": skipped_experiment,
|
|
"attempt": state["attempt"], "decision": "pass", "override": "skip_to_next_stage",
|
|
})
|
|
state["stage_index"] += 1
|
|
state["attempt"] = 0
|
|
state["status"] = "in_progress"
|
|
save_state(state)
|
|
# If this was the last stage, the while loop below never runs
|
|
# (stage_index now == len(STAGES)), so this override's state
|
|
# change would otherwise never get committed/pushed.
|
|
commit_progress(skipped_experiment)
|
|
elif args.force_retry:
|
|
print(f"Human override: retrying stage {state['stage_index'] + 1} "
|
|
f"({STAGES[state['stage_index']]['name']}) after review.")
|
|
state["attempt"] += 1
|
|
state["status"] = "in_progress"
|
|
save_state(state)
|
|
else:
|
|
print(f"BLOCKED at stage {state['stage_index'] + 1} ({STAGES[state['stage_index']]['name']}) "
|
|
f"after {MAX_RETRIES + 1} attempts — see curriculum_state.json's log for eval results.")
|
|
print("Re-run with --force-retry (after adjusting flags/timesteps) or "
|
|
"--skip-to-next-stage (advance anyway) once you've looked at why.")
|
|
sys.exit(1)
|
|
|
|
last_experiment = None
|
|
while state["stage_index"] < len(STAGES):
|
|
stage_index = state["stage_index"]
|
|
attempt = state["attempt"]
|
|
|
|
experiment = run_stage_attempt(stage_index, attempt, args)
|
|
last_experiment = experiment
|
|
reference = reference_bot(stage_index)
|
|
record = evaluate_attempt(experiment, reference, EVAL_EPISODES, stage_index)
|
|
decision = decide(record)
|
|
|
|
print(f"{experiment}: candidate {record['wins_a']}-{record['wins_b']} reference "
|
|
f"({record['draws']} draws) over {record['episodes']} episodes -> {decision}")
|
|
|
|
state["log"].append({
|
|
"stage_index": stage_index, "experiment": experiment, "attempt": attempt,
|
|
"eval": record, "decision": decision,
|
|
})
|
|
|
|
if decision == "pass":
|
|
state["stage_index"] += 1
|
|
state["attempt"] = 0
|
|
state["status"] = "in_progress"
|
|
save_state(state)
|
|
commit_progress(experiment)
|
|
continue
|
|
|
|
if attempt >= MAX_RETRIES:
|
|
state["status"] = "blocked"
|
|
save_state(state)
|
|
commit_progress(experiment)
|
|
print(f"\nBLOCKED: stage {stage_index + 1} ({STAGES[stage_index]['name']}) failed "
|
|
f"{MAX_RETRIES + 1} attempts in a row. Stopping for human review — see "
|
|
f"curriculum_state.json. Re-run with --force-retry or --skip-to-next-stage.")
|
|
sys.exit(1)
|
|
|
|
state["attempt"] += 1
|
|
save_state(state)
|
|
commit_progress(experiment)
|
|
|
|
print("\nCurriculum complete — all stages passed.")
|
|
state["status"] = "done"
|
|
save_state(state)
|
|
if last_experiment is not None:
|
|
# None only if the loop above never ran at all (e.g. re-invoking
|
|
# after the curriculum was already "done") — nothing new to commit
|
|
# in that case.
|
|
commit_progress(last_experiment)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|