Files
CosmicClash/scripts/verify_kind_agones.sh
T
Josh Creek 9ab1bec89a test(kind): dump cluster state before the Agones gate deletes its cluster
This gate fails with nothing but Helm's "context deadline exceeded" and
three Deployments reporting Available: 0/1, then the EXIT trap deletes
the cluster -- so there is no way to learn why the pods never became
ready. Both CI runs and a local run are equally uninformative.

Dump node capacity and conditions, pods and recent events for
agones-system and cosmic-clash, and describe plus current/previous logs
for every not-ready pod, on any failure and before deletion. Events
matter as much as pod status here: FailedScheduling, ImagePullBackOff
and probe failures are all invisible in a status column.

KIND_KEEP_ON_FAILURE=1 retains the cluster for interactive inspection.

Same approach that just found the allocated-Compose cause, where a
silent assertion had hidden a real 422-instead-of-409 API bug across
several CI runs.
2026-09-05 19:57:05 +01:00

161 lines
6.9 KiB
Bash
Executable File

#!/usr/bin/env bash
set -euo pipefail
# Disposable integration gate for multiplayer-next.md §8.49. This deliberately
# does not touch an existing cluster: kind creates an isolated cluster and the
# EXIT trap removes only that named cluster.
root_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
cd "$root_dir"
cluster_name="${KIND_CLUSTER_NAME:-cosmic-clash-agones-smoke}"
agones_version="${AGONES_VERSION:-1.49.0}"
game_server_image="${GAME_SERVER_IMAGE:-cosmic-clash-game-server:kind}"
kind_node_image="${KIND_NODE_IMAGE:-kindest/node:v1.33.1}"
work_dir="$(mktemp -d "${TMPDIR:-/tmp}/cosmic-clash-agones.XXXXXX")"
# This gate fails in CI with nothing but Helm's "context deadline exceeded",
# and the EXIT trap then deletes the cluster, so there is no way to learn why
# the pods never became Available. Dump enough cluster state on failure that a
# CI run explains itself without needing a local reproduction -- which is not
# equivalent anyway, since a developer machine has different resources and a
# different container runtime.
#
# Set KIND_KEEP_ON_FAILURE=1 to retain the cluster for interactive inspection.
on_error() {
echo "kind/Agones gate failed at ${BASH_SOURCE[0]}:$1" >&2
echo "--- failing command: ${BASH_COMMAND}" >&2
}
trap 'on_error "$LINENO"' ERR
dump_cluster_state() {
echo "=== node capacity and conditions ===" >&2
kubectl get nodes -o wide >&2 2>&1 || true
kubectl describe nodes 2>&1 | grep -A 12 -E "Allocated resources|Conditions:" >&2 || true
for ns in agones-system cosmic-clash; do
echo "=== namespace ${ns}: pods ===" >&2
kubectl -n "$ns" get pods -o wide >&2 2>&1 || true
# Events explain scheduling/image/probe failures that pod status alone
# does not: FailedScheduling, ImagePullBackOff, readiness probe errors.
echo "=== namespace ${ns}: recent events ===" >&2
kubectl -n "$ns" get events --sort-by=.lastTimestamp 2>&1 | tail -40 >&2 || true
for pod in $(kubectl -n "$ns" get pods -o jsonpath='{range .items[*]}{.metadata.name}{"\n"}{end}' 2>/dev/null); do
ready="$(kubectl -n "$ns" get pod "$pod" -o jsonpath='{.status.containerStatuses[*].ready}' 2>/dev/null || true)"
case "$ready" in
*false*|"")
echo "=== ${ns}/${pod} is not ready (ready=${ready:-unknown}) ===" >&2
kubectl -n "$ns" describe pod "$pod" 2>&1 | tail -35 >&2 || true
echo "--- ${ns}/${pod} logs (current) ---" >&2
kubectl -n "$ns" logs "$pod" --all-containers --tail=40 >&2 2>&1 || true
echo "--- ${ns}/${pod} logs (previous, if it restarted) ---" >&2
kubectl -n "$ns" logs "$pod" --all-containers --previous --tail=40 >&2 2>&1 || true
;;
esac
done
done
echo "=== helm releases ===" >&2
helm list --all-namespaces >&2 2>&1 || true
}
cleanup() {
local status=$?
if [[ "$status" != 0 ]] && kubectl cluster-info --context "kind-${cluster_name}" >/dev/null 2>&1; then
dump_cluster_state
fi
if [[ "$status" != 0 && "${KIND_KEEP_ON_FAILURE:-}" == 1 ]]; then
echo "kind cluster retained for inspection: kind-${cluster_name} (delete with: kind delete cluster --name ${cluster_name})" >&2
rm -rf "$work_dir"
exit "$status"
fi
kind delete cluster --name "$cluster_name" >/dev/null 2>&1 || true
rm -rf "$work_dir"
exit "$status"
}
trap cleanup EXIT
for tool in docker kind kubectl helm; do
command -v "$tool" >/dev/null 2>&1 || {
echo "8.49 requires '$tool'; install Docker, kind, kubectl, and Helm to run the disposable gate" >&2
exit 2
}
done
if ! docker info >/dev/null 2>&1; then
echo "8.49 requires a running Docker daemon" >&2
exit 2
fi
kind delete cluster --name "$cluster_name" >/dev/null 2>&1 || true
if ! docker image inspect "$game_server_image" >/dev/null 2>&1; then
echo "Building $game_server_image from the pinned game-server target"
docker build --target game-server -t "$game_server_image" .
fi
kind create cluster --name "$cluster_name" --image "$kind_node_image" --wait 120s
kind load docker-image "$game_server_image" --name "$cluster_name"
helm repo add agones https://agones.dev/chart/stable >/dev/null
helm repo update >/dev/null
# Agones 1.49 otherwise requests 10,100 MiB of ephemeral storage for its
# extensions pod, which exceeds a default single-node kind cluster before the
# Fleet can be exercised. These are smoke-only bounds; production resource
# sizing remains deployment-owned.
helm upgrade --install agones agones/agones \
--namespace agones-system --create-namespace \
--version "$agones_version" \
--set agones.crds.cleanup.enabled=true \
--set agones.controller.replicas=1 \
--set agones.extensions.replicas=1 \
--set agones.extensions.resources.requests.ephemeral-storage=128Mi \
--set agones.extensions.resources.limits.ephemeral-storage=512Mi \
--set agones.allocator.replicas=1 \
--wait --timeout 5m
kubectl wait --for=condition=available deployment/agones-controller \
-n agones-system --timeout=180s
kubectl wait --for=condition=available deployment/agones-allocator \
-n agones-system --timeout=180s
# The base Fleet intentionally carries a release-time digest placeholder. For
# this isolated run only, replace that exact placeholder with the image loaded
# into kind. No repository manifest is modified and no mutable image is used
# outside the disposable cluster.
#
# This runner is intentionally an Agones lifecycle smoke, not a substitute for
# the production control-plane gate: there is no PostgreSQL/API/roster backend
# in this disposable cluster. Disable only those production-only child paths so
# the real supervisor can validate the assigned endpoint, launch the exported
# server, and call the Agones SDK Ready endpoint.
zero_digest="$(printf '0%.0s' {1..64})"
sed -e "s|ghcr.io/cosmic-clash/game-server@sha256:${zero_digest}|$game_server_image|" \
-e 's|--control-plane-url=http://control-plane.cosmic-clash.svc.cluster.local:8080|--control-plane-url=|' \
-e '/- --roster-path=\/run\/cosmic-clash\/join-roster.json/d' \
-e '/- --allocated-mode$/d' \
deploy/k8s/base/fleet.yaml > "$work_dir/fleet.yaml"
kubectl apply -f deploy/k8s/base/namespace.yaml
kubectl -n cosmic-clash create secret generic cosmic-clash-game-server \
--from-literal=drain-token=kind-smoke-drain-token \
--from-literal=join-signing-keys.json='{"kind-smoke-key":"a2luZC1zbW9rZS1zaWduaW5nLWtleQ=="}' \
--from-literal=join-signing-key-id=kind-smoke-key \
--dry-run=client -o yaml | kubectl apply -f -
kubectl apply -f deploy/k8s/base/service-accounts.yaml
kubectl apply -f "$work_dir/fleet.yaml"
kubectl wait --for=jsonpath='{.status.ready}'=2 \
fleet/cosmic-clash-game -n cosmic-clash --timeout=5m
cat > "$work_dir/allocation.yaml" <<'EOF'
apiVersion: allocation.agones.dev/v1
kind: GameServerAllocation
metadata:
generateName: cosmic-clash-smoke-
namespace: cosmic-clash
spec:
fleet:
name: cosmic-clash-game
EOF
kubectl create -f "$work_dir/allocation.yaml" -o json > "$work_dir/allocation.json"
python3 scripts/verify_agones_allocation_response.py "$work_dir/allocation.json"