mirror of
https://github.com/jcreek/CosmicClash.git
synced 2026-09-13 12:42:05 +00:00
feat(*): Start curriculum generation 2, seeded from curric-s5-aggression
This commit is contained in:
+133
-83
@@ -18,6 +18,26 @@ forever for the wrong reason. When a stage does fail MAX_RETRIES times in a
|
||||
row, the script stops and asks for a human look rather than retrying
|
||||
indefinitely or silently advancing past a bad stage.
|
||||
|
||||
This is generation 2 of the curriculum. Generation 1 (6 stages: score,
|
||||
defend, no_draws, mechanics, aggression, unmask) ran 2026-07-21 through
|
||||
2026-07-26 and is archived in curriculum_state_gen1.json — its final stage
|
||||
("unmask", full 3D flight on top of the aggression retune) failed 3 straight
|
||||
attempts, monotonically worsening (25% -> 20% -> 15% win rate vs
|
||||
curric-s5-aggression) because every retry resumed the same drifting
|
||||
checkpoint under identical flags instead of actually changing anything.
|
||||
Rather than let generation 1's stage numbering grow indefinitely
|
||||
(unmask-retry4, retry5, ...), generation 2 starts a fresh stage 1 seeded
|
||||
directly from curric-s5-aggression's own checkpoint (FOUNDATION_EXPERIMENT
|
||||
below) — the last stage that actually passed cleanly — carrying over its
|
||||
trained progress without re-running stages 1-5. See TRAINING.md for the
|
||||
full generation 1 history and generation 2's design.
|
||||
|
||||
Every experiment name this script generates is timestamped
|
||||
(YYYYMMDD-HHMM-<name>, applied once in run_stage_attempt) so runs stay
|
||||
unique across restarts/generations and sort chronologically in TensorBoard
|
||||
and checkpoints/ — plain names like "curric-s1-score" from generation 1
|
||||
would otherwise collide with generation 2's own stage 1.
|
||||
|
||||
Usage:
|
||||
.venv/bin/python curriculum.py # run/resume the curriculum
|
||||
.venv/bin/python curriculum.py --seed-checkpoint checkpoints/run11/final.zip
|
||||
@@ -33,12 +53,26 @@ import json
|
||||
import pathlib
|
||||
import subprocess
|
||||
import sys
|
||||
from datetime import datetime
|
||||
|
||||
TRAINING_DIR = pathlib.Path(__file__).resolve().parent
|
||||
STATE_PATH = TRAINING_DIR / "curriculum_state.json"
|
||||
EVAL_HISTORY_PATH = TRAINING_DIR / "eval_history.json"
|
||||
ROOKIE_REFERENCE = TRAINING_DIR.parent / "Game" / "bots" / "rookie.json"
|
||||
|
||||
# Generation 1's last cleanly-passing checkpoint (see curriculum_state_gen1.json)
|
||||
# — generation 2's stage 1 builds on this directly instead of re-running
|
||||
# stages 1-5.
|
||||
FOUNDATION_EXPERIMENT = "curric-s5-aggression"
|
||||
|
||||
# Groundedness (locomotion-mask state) for experiments that predate this
|
||||
# generation's own log, so _grounded_for_experiment can still answer for
|
||||
# them — see that function.
|
||||
LEGACY_GROUNDED = {
|
||||
"rookie": False,
|
||||
FOUNDATION_EXPERIMENT: True,
|
||||
}
|
||||
|
||||
MAX_RETRIES = 2
|
||||
EVAL_EPISODES = 100
|
||||
# "Clear regression" = the reference beats the candidate by at least this
|
||||
@@ -53,82 +87,58 @@ REGRESSION_MARGIN = 0.15
|
||||
STANDING_ARGS = ["--reset-std", "0.3", "--ent-coef", "0.001"]
|
||||
|
||||
STAGES = [
|
||||
{
|
||||
"name": "score",
|
||||
"flags": [
|
||||
"--opponent-mode", "inert",
|
||||
"--attack-goal-bias", "1.0",
|
||||
"--no-allow-vertical", "--no-allow-pitch-roll",
|
||||
],
|
||||
"grounded": True,
|
||||
},
|
||||
{
|
||||
"name": "defend",
|
||||
"flags": [
|
||||
"--opponent-mode", "self_play",
|
||||
"--no-allow-vertical", "--no-allow-pitch-roll",
|
||||
],
|
||||
"grounded": True,
|
||||
},
|
||||
{
|
||||
"name": "no_draws",
|
||||
"flags": ["--draw-penalty", "5"],
|
||||
"grounded": False,
|
||||
},
|
||||
{
|
||||
"name": "mechanics",
|
||||
"flags": [],
|
||||
"grounded": False,
|
||||
},
|
||||
{
|
||||
"name": "aggression",
|
||||
# Deliberately resumes from stage 2 (curric-s2-defend), not stage 4
|
||||
# (see resume_from_experiment/reference_experiment below) — the
|
||||
# locomotion-mask inference bugfix (8c15c46) revealed that stage 3's
|
||||
# full-3D unmask was a clear regression, not an improvement: fairly
|
||||
# evaluated, curric-s2-defend beats both curric-s3-no_draws (26-60)
|
||||
# and curric-s4-mechanics (24-57). Rather than compound that
|
||||
# regression, this stage keeps the locomotion mask ON (matching
|
||||
# stage 2's own regime) and just retunes ball-pursuit reward weights,
|
||||
# so it can't reopen the same grounded-to-3D transition that caused
|
||||
# the earlier failure. Full 3D flight is parked as a separate,
|
||||
# later initiative.
|
||||
"flags": [
|
||||
"--opponent-mode", "self_play",
|
||||
"--no-allow-vertical", "--no-allow-pitch-roll",
|
||||
"--velocity-to-ball-weight", "0.05", # up from 0.02
|
||||
"--ball-distance-penalty", "0.006", # up from 0.002
|
||||
"--ball-touch-reward", "0.5", # up from 0.4
|
||||
],
|
||||
"grounded": True,
|
||||
"resume_from_experiment": "curric-s2-defend",
|
||||
"reference_experiment": "curric-s2-defend",
|
||||
},
|
||||
{
|
||||
"name": "unmask",
|
||||
# Re-opens full 3D controls (no more --no-allow-vertical/
|
||||
# --no-allow-pitch-roll) on top of the aggression retune, instead of
|
||||
# keeping locomotion masked indefinitely. The mask blocked *thrust*-
|
||||
# driven flight outright; the new airborne_penalty (dense, scaled by
|
||||
# height above the floor — see ship_ai_controller.gd) is meant to
|
||||
# teach the policy to prefer staying grounded through incentives
|
||||
# rather than a hard constraint, so it can start learning when the
|
||||
# other axes are actually useful (aerial saves, wall recoveries)
|
||||
# instead of never touching them. This resumes the exact regime
|
||||
# shift (grounded checkpoint -> full 3D) that regressed stage 3 —
|
||||
# the mitigation this time is airborne_penalty plus a much longer
|
||||
# run (24h / ~240M steps vs stage 3's 20M) to actually re-converge
|
||||
# instead of stalling mid-shift like stage 3 did in a fifth of the
|
||||
# time.
|
||||
# driven flight outright; airborne_penalty (dense, scaled by height
|
||||
# above the floor — see ship_ai_controller.gd) is meant to teach the
|
||||
# policy to prefer staying grounded through incentives rather than a
|
||||
# hard constraint, so it can start learning when the other axes are
|
||||
# actually useful (aerial saves, wall recoveries) instead of never
|
||||
# touching them.
|
||||
#
|
||||
# Generation 1 ran this exact transition 3 times (unmask, retry1,
|
||||
# retry2) with identical flags and got monotonically worse each time
|
||||
# (25% -> 20% -> 15% win rate vs curric-s5-aggression) — a blind
|
||||
# retry just continues training the same drifting policy for
|
||||
# longer, it was never going to converge differently. An adversarial
|
||||
# review of a first patch (two modest new flags, still resuming the
|
||||
# drifted retry2 checkpoint) found that insufficient too: the resume
|
||||
# target was the worst of the three already-degraded checkpoints,
|
||||
# and the new weights were too small to compete with the unchanged
|
||||
# ball-pursuit terms. Generation 2's stage 1 instead:
|
||||
# - resumes from FOUNDATION_EXPERIMENT (curric-s5-aggression)
|
||||
# directly (resume_from_experiment below, plus
|
||||
# reset_retry_checkpoint so this stage's own retries reset here
|
||||
# too instead of drifting a failed attempt forward).
|
||||
# - raises velocity_to_ball_weight and ball_distance_penalty
|
||||
# further (the actual ball-chasing terms, unchanged since stage
|
||||
# 5 despite three failed attempts) and ball_touch_reward
|
||||
# alongside them.
|
||||
# - raises ball_velocity_to_goal_weight (reward for moving the
|
||||
# ball toward the goal, not just touching it) and goal_reward
|
||||
# (the terminal reward for scoring) — both newly exposed via
|
||||
# train.py, previously only reachable as raw Godot cmdline args.
|
||||
# - adds draw_penalty (proven effective in generation 1's stage 3
|
||||
# against passivity), which this transition had never set:
|
||||
# previously all carrot for scoring, no stick for never scoring.
|
||||
"flags": [
|
||||
"--opponent-mode", "self_play",
|
||||
"--velocity-to-ball-weight", "0.05",
|
||||
"--ball-distance-penalty", "0.006",
|
||||
"--ball-touch-reward", "0.5",
|
||||
"--velocity-to-ball-weight", "0.08", # up from 0.05
|
||||
"--ball-distance-penalty", "0.01", # up from 0.006
|
||||
"--ball-touch-reward", "0.7", # up from 0.5
|
||||
"--airborne-penalty", "0.003",
|
||||
"--ball-velocity-to-goal-weight", "0.06", # up from 0.02 (0.004 default)
|
||||
"--goal-reward", "80", # up from 60 (40 default)
|
||||
"--draw-penalty", "5",
|
||||
],
|
||||
"grounded": False,
|
||||
"timesteps": 240_000_000, # ~24h at the standing n-parallel/speedup (20M took ~2h)
|
||||
"resume_from_experiment": FOUNDATION_EXPERIMENT,
|
||||
"reference_experiment": FOUNDATION_EXPERIMENT,
|
||||
"reset_retry_checkpoint": True,
|
||||
},
|
||||
]
|
||||
|
||||
@@ -148,23 +158,43 @@ def experiment_name(stage_index: int, attempt: int) -> str:
|
||||
return name if attempt == 0 else f"{name}-retry{attempt}"
|
||||
|
||||
|
||||
def _logged_experiment_name(stage_index: int, attempt: int) -> str:
|
||||
"""The actual (timestamped) experiment name recorded when this attempt
|
||||
ran — needed anywhere a *past* attempt's real name matters, since
|
||||
experiment_name() alone no longer identifies a run on disk (see
|
||||
run_stage_attempt's timestamp prefix)."""
|
||||
state = load_state()
|
||||
for entry in state["log"]:
|
||||
if entry["stage_index"] == stage_index and entry["attempt"] == attempt:
|
||||
return entry["experiment"]
|
||||
raise RuntimeError(f"No logged experiment for stage {stage_index} attempt {attempt}")
|
||||
|
||||
|
||||
def resume_checkpoint(stage_index: int, attempt: int, seed_checkpoint: str | None) -> str | None:
|
||||
if attempt > 0:
|
||||
if attempt > 0 and not STAGES[stage_index].get("reset_retry_checkpoint"):
|
||||
# Retry: keep training the same stage's own last attempt.
|
||||
prev = experiment_name(stage_index, attempt - 1)
|
||||
prev = _logged_experiment_name(stage_index, attempt - 1)
|
||||
return str(TRAINING_DIR / "checkpoints" / prev / "final.zip")
|
||||
if stage_index == 0:
|
||||
if stage_index == 0 and seed_checkpoint:
|
||||
return seed_checkpoint
|
||||
if stage_index == 0 and not STAGES[0].get("resume_from_experiment"):
|
||||
# Deliberately fresh by default: the curriculum exists because
|
||||
# resuming self-play across a regime change (run10, run11) didn't
|
||||
# work, so stage 1 starts from a random policy under its own
|
||||
# regime unless --seed-checkpoint says otherwise.
|
||||
return seed_checkpoint
|
||||
# work, so a from-scratch stage 1 starts from a random policy under
|
||||
# its own regime unless --seed-checkpoint or resume_from_experiment
|
||||
# says otherwise.
|
||||
return None
|
||||
# Either a later stage chaining off its predecessor, or
|
||||
# reset_retry_checkpoint: this stage's own retries have been drifting
|
||||
# rather than converging (see the "unmask" stage's comment) — resume
|
||||
# from the stage's normal resume source instead of compounding the last
|
||||
# failed attempt's drift.
|
||||
prev_experiment = _resume_source_experiment(stage_index)
|
||||
return str(TRAINING_DIR / "checkpoints" / prev_experiment / "final.zip")
|
||||
|
||||
|
||||
def reference_bot(stage_index: int) -> str:
|
||||
if stage_index == 0:
|
||||
if stage_index == 0 and not STAGES[0].get("reference_experiment"):
|
||||
return str(ROOKIE_REFERENCE)
|
||||
prev_experiment = _reference_source_experiment(stage_index)
|
||||
return str(TRAINING_DIR.parent / "Game" / "bots" / f"{prev_experiment}.json")
|
||||
@@ -172,8 +202,8 @@ def reference_bot(stage_index: int) -> str:
|
||||
|
||||
# A stage normally chains off "whatever passed at the previous index," but a
|
||||
# stage can instead name an explicit resume_from_experiment/reference_experiment
|
||||
# to skip a since-regressed branch (see the "aggression" stage) without
|
||||
# rewriting history for the stages it's skipping past.
|
||||
# to skip a since-regressed branch, or (stage 0) to seed from a fixed
|
||||
# foundation checkpoint instead of a from-scratch policy.
|
||||
def _resume_source_experiment(stage_index: int) -> str:
|
||||
override = STAGES[stage_index].get("resume_from_experiment")
|
||||
return override if override else _passing_experiment_for_stage(stage_index - 1)
|
||||
@@ -193,16 +223,19 @@ def _passing_experiment_for_stage(stage_index: int) -> str:
|
||||
|
||||
|
||||
def _grounded_for_experiment(experiment: str) -> bool:
|
||||
if experiment == "rookie":
|
||||
return False
|
||||
for index, stage in enumerate(STAGES):
|
||||
if experiment_name(index, 0) == experiment:
|
||||
return stage["grounded"]
|
||||
if experiment in LEGACY_GROUNDED:
|
||||
return LEGACY_GROUNDED[experiment]
|
||||
state = load_state()
|
||||
for entry in state["log"]:
|
||||
if entry["experiment"] == experiment:
|
||||
return STAGES[entry["stage_index"]]["grounded"]
|
||||
raise ValueError(f"Unknown experiment for groundedness lookup: {experiment}")
|
||||
|
||||
|
||||
def run_stage_attempt(stage_index: int, attempt: int, args) -> str:
|
||||
exp = experiment_name(stage_index, attempt)
|
||||
# Timestamped so names stay unique across restarts/generations and sort
|
||||
# chronologically in TensorBoard/checkpoints — see module docstring.
|
||||
exp = f"{datetime.now().strftime('%Y%m%d-%H%M')}-{experiment_name(stage_index, attempt)}"
|
||||
resume = resume_checkpoint(stage_index, attempt, args.seed_checkpoint)
|
||||
# A stage can override the run's timesteps budget (see "floor-lock",
|
||||
# which deliberately runs much longer than the ~20M/~2h every stage so
|
||||
@@ -226,7 +259,7 @@ def run_stage_attempt(stage_index: int, attempt: int, args) -> str:
|
||||
|
||||
|
||||
def reference_grounded(stage_index: int) -> bool:
|
||||
if stage_index == 0:
|
||||
if stage_index == 0 and not STAGES[0].get("reference_experiment"):
|
||||
# rookie.json predates the locomotion mask entirely — always full 3D.
|
||||
return False
|
||||
return _grounded_for_experiment(_reference_source_experiment(stage_index))
|
||||
@@ -272,7 +305,11 @@ def main():
|
||||
parser.add_argument("--timesteps", type=int, default=20_000_000)
|
||||
parser.add_argument("--n-parallel", type=int, default=14)
|
||||
parser.add_argument("--speedup", type=int, default=16)
|
||||
parser.add_argument("--seed-checkpoint", default=None, help="Resume stage 1 from this checkpoint instead of a fresh policy")
|
||||
parser.add_argument(
|
||||
"--seed-checkpoint", default=None,
|
||||
help="Resume stage 1 from this checkpoint instead of its default resume source "
|
||||
"(FOUNDATION_EXPERIMENT's checkpoint)",
|
||||
)
|
||||
parser.add_argument("--force-retry", action="store_true", help="Retry a blocked stage after human review")
|
||||
parser.add_argument("--skip-to-next-stage", action="store_true", help="Human judgment call: treat the blocked stage as good enough, advance anyway")
|
||||
args = parser.parse_args()
|
||||
@@ -283,14 +320,20 @@ def main():
|
||||
if args.skip_to_next_stage:
|
||||
print(f"Human override: advancing past stage {state['stage_index'] + 1} "
|
||||
f"({STAGES[state['stage_index']]['name']}) despite exhausted retries.")
|
||||
skipped_experiment = _logged_experiment_name(state["stage_index"], state["attempt"])
|
||||
state["log"].append({
|
||||
"stage_index": state["stage_index"], "experiment": experiment_name(state["stage_index"], state["attempt"]),
|
||||
"stage_index": state["stage_index"],
|
||||
"experiment": skipped_experiment,
|
||||
"attempt": state["attempt"], "decision": "pass", "override": "skip_to_next_stage",
|
||||
})
|
||||
state["stage_index"] += 1
|
||||
state["attempt"] = 0
|
||||
state["status"] = "in_progress"
|
||||
save_state(state)
|
||||
# If this was the last stage, the while loop below never runs
|
||||
# (stage_index now == len(STAGES)), so this override's state
|
||||
# change would otherwise never get committed/pushed.
|
||||
commit_progress(skipped_experiment)
|
||||
elif args.force_retry:
|
||||
print(f"Human override: retrying stage {state['stage_index'] + 1} "
|
||||
f"({STAGES[state['stage_index']]['name']}) after review.")
|
||||
@@ -304,11 +347,13 @@ def main():
|
||||
"--skip-to-next-stage (advance anyway) once you've looked at why.")
|
||||
sys.exit(1)
|
||||
|
||||
last_experiment = None
|
||||
while state["stage_index"] < len(STAGES):
|
||||
stage_index = state["stage_index"]
|
||||
attempt = state["attempt"]
|
||||
|
||||
experiment = run_stage_attempt(stage_index, attempt, args)
|
||||
last_experiment = experiment
|
||||
reference = reference_bot(stage_index)
|
||||
record = evaluate_attempt(experiment, reference, EVAL_EPISODES, stage_index)
|
||||
decision = decide(record)
|
||||
@@ -345,6 +390,11 @@ def main():
|
||||
print("\nCurriculum complete — all stages passed.")
|
||||
state["status"] = "done"
|
||||
save_state(state)
|
||||
if last_experiment is not None:
|
||||
# None only if the loop above never ran at all (e.g. re-invoking
|
||||
# after the curriculum was already "done") — nothing new to commit
|
||||
# in that case.
|
||||
commit_progress(last_experiment)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -1,137 +1,6 @@
|
||||
{
|
||||
"stage_index": 5,
|
||||
"attempt": 2,
|
||||
"status": "blocked",
|
||||
"log": [
|
||||
{
|
||||
"stage_index": 0,
|
||||
"experiment": "curric-s1-score",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-21T15:52:20+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s1-score.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/rookie.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 22,
|
||||
"wins_b": 16,
|
||||
"draws": 62,
|
||||
"win_rate_a": 0.22
|
||||
},
|
||||
"decision": "pass"
|
||||
},
|
||||
{
|
||||
"stage_index": 1,
|
||||
"experiment": "curric-s2-defend",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-21T18:19:44+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s2-defend.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s1-score.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 23,
|
||||
"wins_b": 16,
|
||||
"draws": 61,
|
||||
"win_rate_a": 0.23
|
||||
},
|
||||
"decision": "pass"
|
||||
},
|
||||
{
|
||||
"stage_index": 2,
|
||||
"experiment": "curric-s3-no_draws",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-22T11:36:42+00:00",
|
||||
"model_a": "/Users/jcreek/Documents/repos/GitHub/CosmicClash/Game/bots/curric-s3-no_draws.json",
|
||||
"model_b": "/Users/jcreek/Documents/repos/GitHub/CosmicClash/Game/bots/curric-s2-defend.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 26,
|
||||
"wins_b": 60,
|
||||
"draws": 14,
|
||||
"win_rate_a": 0.26
|
||||
},
|
||||
"decision": "fail",
|
||||
"note": "Original eval (44-19, recorded 2026-07-21T20:46:34) predates the locomotion-mask inference bugfix (8c15c46) and ran with the grounded stage-2 reference unfairly unmasked. Re-run post-fix with --grounded-b reverses the verdict: stage 3's full-3D unmask is a clear regression from stage 2, not an improvement. Not retried via the normal flag-retry mechanism \u2014 see stage_index 4 (aggression), which redirects around this branch by resuming from curric-s2-defend directly instead."
|
||||
},
|
||||
{
|
||||
"stage_index": 3,
|
||||
"experiment": "curric-s4-mechanics",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-22T11:38:57+00:00",
|
||||
"model_a": "/Users/jcreek/Documents/repos/GitHub/CosmicClash/Game/bots/curric-s4-mechanics.json",
|
||||
"model_b": "/Users/jcreek/Documents/repos/GitHub/CosmicClash/Game/bots/curric-s3-no_draws.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 26,
|
||||
"wins_b": 33,
|
||||
"draws": 41,
|
||||
"win_rate_a": 0.26
|
||||
},
|
||||
"decision": "pass",
|
||||
"note": "Passes only against its own (already-regressed) predecessor, curric-s3-no_draws. Evaluated directly against grounded curric-s2-defend (2026-07-22T11:40:07), curric-s4-mechanics also loses clearly: 24-57-19. Do not treat this stage's 'pass' as evidence curric-s4-mechanics is the strongest available model overall \u2014 see stage 2's note and stage_index 4 (aggression)."
|
||||
},
|
||||
{
|
||||
"stage_index": 4,
|
||||
"experiment": "curric-s5-aggression",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-22T19:33:07+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s5-aggression.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s2-defend.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 41,
|
||||
"wins_b": 47,
|
||||
"draws": 12,
|
||||
"win_rate_a": 0.41
|
||||
},
|
||||
"decision": "pass"
|
||||
},
|
||||
{
|
||||
"stage_index": 5,
|
||||
"experiment": "curric-s6-unmask",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-24T01:25:37+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s6-unmask.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s5-aggression.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 25,
|
||||
"wins_b": 50,
|
||||
"draws": 25,
|
||||
"win_rate_a": 0.25
|
||||
},
|
||||
"decision": "fail"
|
||||
},
|
||||
{
|
||||
"stage_index": 5,
|
||||
"experiment": "curric-s6-unmask-retry1",
|
||||
"attempt": 1,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-25T06:41:43+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s6-unmask-retry1.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s5-aggression.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 20,
|
||||
"wins_b": 61,
|
||||
"draws": 19,
|
||||
"win_rate_a": 0.2
|
||||
},
|
||||
"decision": "fail"
|
||||
},
|
||||
{
|
||||
"stage_index": 5,
|
||||
"experiment": "curric-s6-unmask-retry2",
|
||||
"attempt": 2,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-26T12:08:35+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s6-unmask-retry2.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s5-aggression.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 15,
|
||||
"wins_b": 68,
|
||||
"draws": 17,
|
||||
"win_rate_a": 0.15
|
||||
},
|
||||
"decision": "fail"
|
||||
}
|
||||
]
|
||||
"stage_index": 0,
|
||||
"attempt": 0,
|
||||
"status": "in_progress",
|
||||
"log": []
|
||||
}
|
||||
|
||||
@@ -0,0 +1,137 @@
|
||||
{
|
||||
"stage_index": 5,
|
||||
"attempt": 2,
|
||||
"status": "blocked",
|
||||
"log": [
|
||||
{
|
||||
"stage_index": 0,
|
||||
"experiment": "curric-s1-score",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-21T15:52:20+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s1-score.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/rookie.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 22,
|
||||
"wins_b": 16,
|
||||
"draws": 62,
|
||||
"win_rate_a": 0.22
|
||||
},
|
||||
"decision": "pass"
|
||||
},
|
||||
{
|
||||
"stage_index": 1,
|
||||
"experiment": "curric-s2-defend",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-21T18:19:44+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s2-defend.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s1-score.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 23,
|
||||
"wins_b": 16,
|
||||
"draws": 61,
|
||||
"win_rate_a": 0.23
|
||||
},
|
||||
"decision": "pass"
|
||||
},
|
||||
{
|
||||
"stage_index": 2,
|
||||
"experiment": "curric-s3-no_draws",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-22T11:36:42+00:00",
|
||||
"model_a": "/Users/jcreek/Documents/repos/GitHub/CosmicClash/Game/bots/curric-s3-no_draws.json",
|
||||
"model_b": "/Users/jcreek/Documents/repos/GitHub/CosmicClash/Game/bots/curric-s2-defend.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 26,
|
||||
"wins_b": 60,
|
||||
"draws": 14,
|
||||
"win_rate_a": 0.26
|
||||
},
|
||||
"decision": "fail",
|
||||
"note": "Original eval (44-19, recorded 2026-07-21T20:46:34) predates the locomotion-mask inference bugfix (8c15c46) and ran with the grounded stage-2 reference unfairly unmasked. Re-run post-fix with --grounded-b reverses the verdict: stage 3's full-3D unmask is a clear regression from stage 2, not an improvement. Not retried via the normal flag-retry mechanism \u2014 see stage_index 4 (aggression), which redirects around this branch by resuming from curric-s2-defend directly instead."
|
||||
},
|
||||
{
|
||||
"stage_index": 3,
|
||||
"experiment": "curric-s4-mechanics",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-22T11:38:57+00:00",
|
||||
"model_a": "/Users/jcreek/Documents/repos/GitHub/CosmicClash/Game/bots/curric-s4-mechanics.json",
|
||||
"model_b": "/Users/jcreek/Documents/repos/GitHub/CosmicClash/Game/bots/curric-s3-no_draws.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 26,
|
||||
"wins_b": 33,
|
||||
"draws": 41,
|
||||
"win_rate_a": 0.26
|
||||
},
|
||||
"decision": "pass",
|
||||
"note": "Passes only against its own (already-regressed) predecessor, curric-s3-no_draws. Evaluated directly against grounded curric-s2-defend (2026-07-22T11:40:07), curric-s4-mechanics also loses clearly: 24-57-19. Do not treat this stage's 'pass' as evidence curric-s4-mechanics is the strongest available model overall \u2014 see stage 2's note and stage_index 4 (aggression)."
|
||||
},
|
||||
{
|
||||
"stage_index": 4,
|
||||
"experiment": "curric-s5-aggression",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-22T19:33:07+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s5-aggression.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s2-defend.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 41,
|
||||
"wins_b": 47,
|
||||
"draws": 12,
|
||||
"win_rate_a": 0.41
|
||||
},
|
||||
"decision": "pass"
|
||||
},
|
||||
{
|
||||
"stage_index": 5,
|
||||
"experiment": "curric-s6-unmask",
|
||||
"attempt": 0,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-24T01:25:37+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s6-unmask.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s5-aggression.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 25,
|
||||
"wins_b": 50,
|
||||
"draws": 25,
|
||||
"win_rate_a": 0.25
|
||||
},
|
||||
"decision": "fail"
|
||||
},
|
||||
{
|
||||
"stage_index": 5,
|
||||
"experiment": "curric-s6-unmask-retry1",
|
||||
"attempt": 1,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-25T06:41:43+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s6-unmask-retry1.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s5-aggression.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 20,
|
||||
"wins_b": 61,
|
||||
"draws": 19,
|
||||
"win_rate_a": 0.2
|
||||
},
|
||||
"decision": "fail"
|
||||
},
|
||||
{
|
||||
"stage_index": 5,
|
||||
"experiment": "curric-s6-unmask-retry2",
|
||||
"attempt": 2,
|
||||
"eval": {
|
||||
"timestamp": "2026-07-26T12:08:35+00:00",
|
||||
"model_a": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s6-unmask-retry2.json",
|
||||
"model_b": "/home/jcreek/ai-training/CosmicClash/Game/bots/curric-s5-aggression.json",
|
||||
"episodes": 100,
|
||||
"wins_a": 15,
|
||||
"wins_b": 68,
|
||||
"draws": 17,
|
||||
"win_rate_a": 0.15
|
||||
},
|
||||
"decision": "fail"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -100,6 +100,14 @@ def parse_args():
|
||||
"--airborne-penalty", type=float, default=None,
|
||||
help="Overrides ShipAIController.airborne_penalty (dense per-tick cost scaled by height above the floor)",
|
||||
)
|
||||
curriculum.add_argument(
|
||||
"--ball-velocity-to-goal-weight", type=float, default=None,
|
||||
help="Overrides ShipAIController.ball_velocity_to_goal_weight (dense reward for the ball's velocity toward the attack goal)",
|
||||
)
|
||||
curriculum.add_argument(
|
||||
"--goal-reward", type=float, default=None,
|
||||
help="Overrides TrainingMode.goal_reward (terminal reward for actually scoring)",
|
||||
)
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
@@ -121,6 +129,8 @@ def _curriculum_kwargs(args) -> dict:
|
||||
"ai_ball_distance_penalty": args.ball_distance_penalty,
|
||||
"ai_ball_touch_reward": args.ball_touch_reward,
|
||||
"ai_airborne_penalty": args.airborne_penalty,
|
||||
"ai_ball_velocity_to_goal_weight": args.ball_velocity_to_goal_weight,
|
||||
"goal_reward": args.goal_reward,
|
||||
}
|
||||
return {key: value for key, value in mapping.items() if value is not None}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user