diff --git a/show-status.sh b/show-status.sh index ef68d945..ce026a24 100755 --- a/show-status.sh +++ b/show-status.sh @@ -22,26 +22,25 @@ check_sync_status() { # Cap the whole per-node branch (belt-and-suspenders over check-health's own cap), so no single # node can ever block the 'wait' below — that is what wedged the fleet rpc-update for hours. result=$(timeout "${SYNC_TIMEOUT:-60}" "$BASEPATH/sync-status.sh" "${part%.yml}") + # Capture the status IMMEDIATELY. Any command in between - including a plain + # assignment like `code=0` - overwrites $? with its own (always 0) status. + rc=$? code=0 - - if [ $? -ne 0 ]; then - if [[ "$result" == *"syncing"* ]]; then - # Allow exit status 1 if result contains "syncing" - code=0 - elif [[ "$result" == *"lagging"* ]]; then - # Allow exit status 1 if result contains "lagging" + if [ "$rc" -ne 0 ]; then + if [[ "$result" == *"syncing"* ]] || [[ "$result" == *"lagging"* ]]; then + # sync-status exits 1 for syncing/lagging; those are expected states, + # not failures. code=0 else - any_failure=true code=1 fi - else - code=1 - any_failure=true fi echo "${part%.yml}: $result" + # NOTE: do NOT set any_failure here. This function runs backgrounded (`&`), so + # it executes in a subshell and any variable it sets is discarded. Failure is + # propagated to the parent through this return code, collected by `wait` below. return "$code" } @@ -74,9 +73,12 @@ for part in "${parts[@]}"; do fi done -# Wait for all background processes to finish +# Wait for all background processes to finish. `wait` runs in the PARENT shell, so +# this is where a failing node can actually flip any_failure - the checker itself +# cannot, being a subshell. Previously the status was discarded here, which silently +# neutered the exit code. for pid in "${pids[@]}"; do - wait "$pid" + wait "$pid" || any_failure=true done # Fenced nodes (fleet-state maintenance windows) are dropped from COMPOSE_FILE