fix(deploy): let EB health converge before failing the version gate

Elastic Beanstalk reports Ready as soon as a rollout finishes, before
enhanced health has converged. Both verification gates decided on the first
Ready poll, so a release whose version had activated correctly was failed on
a health value that had not settled yet — and then rolled back.

The failure message compounded it: run 34293894914 printed 'Environment
became Ready without activating expected version a0fdd199...' when the
active version was exactly a0fdd199... The discriminator was health, not
version, which sends whoever reads the log after the wrong problem.

Separate the two conditions, keep polling while the correct version is
active but health has not settled, and report the last observed state on
timeout. An environment that stays unhealthy for the full window still
fails; this does not widen what counts as a good deploy.
This commit is contained in:
Alexandre Brandizzi 2026-09-08 21:31:55 -03:00
parent a0fdd19934
commit d1e3fb03ce

View file

@ -330,17 +330,25 @@ jobs:
echo "environment status: $status; version: $current; health: $health"
if [ "$status" = "Ready" ]; then
if [ "$current" = "$expected" ] && { [ "$health" = "Green" ] || [ "$health" = "Yellow" ]; }; then
if [ "$current" != "$expected" ]; then
echo "Environment became Ready on version $current, not the expected $expected." >&2
exit 1
fi
if [ "$health" = "Green" ] || [ "$health" = "Yellow" ]; then
echo "Expected application version is Ready and healthy."
exit 0
fi
echo "Environment became Ready without activating expected version $expected." >&2
exit 1
# The expected version IS active. Elastic Beanstalk reports Ready as
# soon as the rollout finishes, before enhanced health has converged,
# so deciding on the first Ready poll fails a good release on a health
# value that was always going to change. Keep polling; an environment
# that is genuinely unhealthy still fails when the window runs out.
echo "Expected version is active; waiting for health to leave $health."
fi
sleep 15
done
echo "Expected application version did not become Ready within the deployment window." >&2
echo "Expected application version did not become Ready and healthy within the deployment window (last seen: status=$status version=$current health=$health)." >&2
exit 1
- name: Post-deploy smoke
@ -724,17 +732,25 @@ jobs:
echo "environment status: $status; version: $current; health: $health"
if [ "$status" = "Ready" ]; then
if [ "$current" = "$expected" ] && { [ "$health" = "Green" ] || [ "$health" = "Yellow" ]; }; then
if [ "$current" != "$expected" ]; then
echo "Environment became Ready on version $current, not the expected $expected." >&2
exit 1
fi
if [ "$health" = "Green" ] || [ "$health" = "Yellow" ]; then
echo "Expected application version is Ready and healthy."
exit 0
fi
echo "Environment became Ready without activating expected version $expected." >&2
exit 1
# The expected version IS active. Elastic Beanstalk reports Ready as
# soon as the rollout finishes, before enhanced health has converged,
# so deciding on the first Ready poll fails a good release on a health
# value that was always going to change. Keep polling; an environment
# that is genuinely unhealthy still fails when the window runs out.
echo "Expected version is active; waiting for health to leave $health."
fi
sleep 15
done
echo "Expected application version did not become Ready within the deployment window." >&2
echo "Expected application version did not become Ready and healthy within the deployment window (last seen: status=$status version=$current health=$health)." >&2
exit 1
- name: Post-deploy smoke
@ -822,15 +838,21 @@ jobs:
)
echo "environment status: $status; version: $current; health: $health"
if [ "$status" = "Ready" ]; then
if [ "$current" = "$prev" ] && { [ "$health" = "Green" ] || [ "$health" = "Yellow" ]; }; then
if [ "$current" != "$prev" ]; then
echo "Rollback reached Ready on version $current, not the previous $prev." >&2
exit 1
fi
if [ "$health" = "Green" ] || [ "$health" = "Yellow" ]; then
echo "Application version restore complete; previous code is Ready and healthy."
exit 0
fi
echo "Rollback reached Ready in an unexpected version/health state." >&2
exit 1
# Same convergence gap as the release check above: the previous version
# is back, health has not settled yet, and reporting a failed rollback
# here hides the fact that the restore itself worked.
echo "Previous version is active; waiting for health to leave $health."
fi
sleep 15
done
echo "Environment did not return to Ready within rollback window." >&2
echo "Environment did not return to Ready and healthy within the rollback window (last seen: status=$status version=$current health=$health)." >&2
exit 1