From ee8c389bdb464df96c6b4a2bbd5192b779db0f96 Mon Sep 17 00:00:00 2001 From: agenticode <16611333+agenticode@users.noreply.github.com> Date: Wed, 26 Aug 2026 21:52:46 +0900 Subject: [PATCH] test(e2e): stop racing the controller's own drain budget 4b waited 60x5s = 5m for a node to disappear. The actuator's WaitNodeEmpty gives the drain NodeDrainTimeout (default 5m) before it gives up, and only the next reconcile retries. The two budgets were identical, so any drain that was not near-instant became a coin flip: the test killed the controller mid-drain and blamed it for "never removed a node". Symptom in CI: ~1 run in 3 failed, always at 4b, always with step 3 delete-node err="context canceled" at the exact moment of teardown. Failing runs sat the full 7m20s; green runs finished the whole script in ~2m10s. Wait 10m instead, which covers a worst-case drain plus one retry, log progress so the wait is not a silent stall, and dump node/pod state on timeout so a real failure says why. Co-authored-by: kording <74226694+kording@users.noreply.github.com> --- test/e2e/e2e.sh | 21 +++++++++++++++++++-- 1 file changed, 19 insertions(+), 2 deletions(-) diff --git a/test/e2e/e2e.sh b/test/e2e/e2e.sh index 28c6b87..d76622a 100755 --- a/test/e2e/e2e.sh +++ b/test/e2e/e2e.sh @@ -135,11 +135,28 @@ KC get node "$CLUSTER-worker3" >/dev/null || FAIL "emergency drain must NOT dele PASS "spot interruption: worker3 cordoned and drained, node left for cloud reclamation" echo "==> 4b) consolidation of underutilized on-demand workers" -for i in $(seq 1 60); do +# Patience must exceed the controller's own drain budget, or the test and the +# thing it tests race each other. The actuator gives WaitNodeEmpty +# NodeDrainTimeout (default 5m) before it reports the drain as failed, and only +# the *next* reconcile (--interval above) retries. A 5m wait here therefore +# ties with the controller's 5m and the test always loses: it kills the +# controller mid-drain and reports "never removed a node" for what is really a +# slow-but-healthy drain. Seen as a ~1-in-3 CI flake, always the same symptom, +# with the failing runs sitting the full budget while green runs finish in ~2m. +# 10m covers a worst-case drain plus one retry. +for i in $(seq 1 120); do NODES_NOW=$(KC get nodes --no-headers | wc -l | tr -d ' ') [[ "$NODES_NOW" -lt "$NODES_BEFORE" ]] && break + (( i % 12 == 0 )) && echo " still $NODES_NOW nodes after $((i * 5))s..." sleep 5 - [[ $i == 60 ]] && FAIL "controller never removed a node" + if [[ $i == 120 ]]; then + # A bare "never removed a node" says nothing about *why*. Dump the state + # the next reader will want: who is cordoned, and what still pins the node. + echo "--- nodes ---" >&2; KC get nodes -o wide >&2 || true + echo "--- non-DaemonSet pods by node ---" >&2 + KC get pods -A -o wide --no-headers >&2 || true + FAIL "controller never removed a node" + fi done PASS "node removed: $NODES_BEFORE → $NODES_NOW" KC get node "$CLUSTER-worker3" >/dev/null 2>&1 || FAIL "karpenter-managed node was consolidated"