mirror of
https://github.com/jcreek/CosmicClash.git
synced 2026-09-10 16:04:04 +00:00
8c15c466ef
AIShipController (eval + real gameplay) ran the raw policy output unmasked regardless of allow_vertical/allow_pitch_roll, while ShipAIController (training) correctly discarded those axes for grounded curriculum stages. A grounded-trained model's untrained vertical/pitch-roll output reached the ship as noise during eval, understating it against models that were never handicapped this way.
123 lines
4.9 KiB
Python
123 lines
4.9 KiB
Python
"""Pit two exported policies against each other and record the result.
|
|
|
|
Uses the same in-Godot inference path that ships in the game
|
|
(AIShipController + PolicyNetwork), so eval strength = in-game strength.
|
|
Episodes are golden-goal: first goal wins, timeout is a draw. Half the
|
|
episodes are played with sides swapped for fairness. Results are appended to
|
|
eval_history.json — the bot-progress-over-time record.
|
|
|
|
Example:
|
|
.venv/bin/python evaluate.py ../Game/bots/rookie.json checkpoints/run01/candidate.json --episodes 40
|
|
"""
|
|
|
|
import argparse
|
|
import datetime
|
|
import json
|
|
import os
|
|
import pathlib
|
|
import subprocess
|
|
|
|
TRAINING_DIR = pathlib.Path(__file__).resolve().parent
|
|
GAME_DIR = TRAINING_DIR.parent / "Game"
|
|
TRAINING_SCENE = "res://scenes/training.tscn"
|
|
DEFAULT_GODOT_MACOS = "/Applications/Godot.app/Contents/MacOS/Godot"
|
|
|
|
|
|
def run_half(
|
|
godot_bin: str,
|
|
model_a: str,
|
|
model_b: str,
|
|
episodes: int,
|
|
speedup: int,
|
|
seed: int,
|
|
grounded_a: bool = False,
|
|
grounded_b: bool = False,
|
|
) -> dict:
|
|
cmd = [
|
|
godot_bin,
|
|
"--path",
|
|
str(GAME_DIR),
|
|
TRAINING_SCENE,
|
|
"--headless",
|
|
"--disable-render-loop",
|
|
f"--eval_model_a={model_a}",
|
|
f"--eval_model_b={model_b}",
|
|
f"--eval_episodes={episodes}",
|
|
f"--speedup={speedup}",
|
|
f"--env_seed={seed}",
|
|
]
|
|
# Must match how each model was actually trained (see AIShipController's
|
|
# allow_vertical/allow_pitch_roll) — a curriculum stage 1/2 model never
|
|
# got a reward gradient on these axes, so leaving them unmasked here adds
|
|
# untrained aerial noise the model's own training never had to contend with.
|
|
if grounded_a:
|
|
cmd += ["--eval_allow_vertical_a=false", "--eval_allow_pitch_roll_a=false"]
|
|
if grounded_b:
|
|
cmd += ["--eval_allow_vertical_b=false", "--eval_allow_pitch_roll_b=false"]
|
|
timeout = episodes * 30 / speedup * 3 + 120 # worst case: all draws, plus margin
|
|
result = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout)
|
|
for line in result.stdout.splitlines():
|
|
if line.startswith("EVAL_RESULT "):
|
|
return json.loads(line[len("EVAL_RESULT "):])
|
|
raise RuntimeError(f"No EVAL_RESULT in godot output:\n{result.stdout[-2000:]}\n{result.stderr[-2000:]}")
|
|
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("model_a", help="Path to first exported policy .json")
|
|
parser.add_argument("model_b", help="Path to second exported policy .json")
|
|
parser.add_argument("--episodes", type=int, default=20, help="Total episodes (split across side swap)")
|
|
parser.add_argument(
|
|
"--godot_bin",
|
|
default=os.environ.get("GODOT_BIN", DEFAULT_GODOT_MACOS),
|
|
help="Path to the Godot binary (or set GODOT_BIN)",
|
|
)
|
|
parser.add_argument("--speedup", type=int, default=16)
|
|
parser.add_argument("--history", default=str(TRAINING_DIR / "eval_history.json"))
|
|
parser.add_argument(
|
|
"--grounded-a", action="store_true", help="model_a was trained with locomotion masked (curriculum stages 1-2)"
|
|
)
|
|
parser.add_argument(
|
|
"--grounded-b", action="store_true", help="model_b was trained with locomotion masked (curriculum stages 1-2)"
|
|
)
|
|
args = parser.parse_args()
|
|
|
|
model_a = str(pathlib.Path(args.model_a).resolve())
|
|
model_b = str(pathlib.Path(args.model_b).resolve())
|
|
half = max(args.episodes // 2, 1)
|
|
|
|
# Half the episodes on each side to cancel any residual side asymmetry;
|
|
# different seeds so the halves see different randomized episode states.
|
|
# Groundedness is per physical model, so it swaps sides along with it.
|
|
first = run_half(args.godot_bin, model_a, model_b, half, args.speedup, seed=1,
|
|
grounded_a=args.grounded_a, grounded_b=args.grounded_b)
|
|
second = run_half(args.godot_bin, model_b, model_a, half, args.speedup, seed=2,
|
|
grounded_a=args.grounded_b, grounded_b=args.grounded_a)
|
|
|
|
record = {
|
|
"timestamp": datetime.datetime.now(datetime.timezone.utc).isoformat(timespec="seconds"),
|
|
"model_a": model_a,
|
|
"model_b": model_b,
|
|
"episodes": first["episodes"] + second["episodes"],
|
|
"wins_a": first["goals_a"] + second["goals_b"],
|
|
"wins_b": first["goals_b"] + second["goals_a"],
|
|
"draws": first["draws"] + second["draws"],
|
|
}
|
|
record["win_rate_a"] = round(record["wins_a"] / record["episodes"], 3)
|
|
|
|
history_path = pathlib.Path(args.history)
|
|
history = json.loads(history_path.read_text()) if history_path.exists() else []
|
|
history.append(record)
|
|
history_path.write_text(json.dumps(history, indent=2) + "\n")
|
|
|
|
print(
|
|
f"{pathlib.Path(model_a).name} vs {pathlib.Path(model_b).name} over {record['episodes']} episodes: "
|
|
f"{record['wins_a']}-{record['wins_b']} ({record['draws']} draws), "
|
|
f"win rate A = {record['win_rate_a']:.0%}"
|
|
)
|
|
print(f"Appended to {history_path}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|