Compare commits

..

1 Commits

3 changed files with 15 additions and 17 deletions

View File

@@ -35,7 +35,7 @@ services:
dockerfile: cometbft.Dockerfile dockerfile: cometbft.Dockerfile
args: args:
CL_IMAGE: ${COSMOS_GAIAD_IMAGE:-ghcr.io/cosmos/gaia} CL_IMAGE: ${COSMOS_GAIAD_IMAGE:-ghcr.io/cosmos/gaia}
CL_VERSION: ${COSMOS_MAINNET_GAIAD_VERSION:-v27.6.0} CL_VERSION: ${COSMOS_MAINNET_GAIAD_VERSION:-v27.5.0}
sysctls: sysctls:
# TCP Performance # TCP Performance
net.ipv4.tcp_slow_start_after_idle: 0 # Disable slow start after idle net.ipv4.tcp_slow_start_after_idle: 0 # Disable slow start after idle

View File

@@ -40,7 +40,7 @@ services:
- chains - chains
labels: labels:
- "traefik.enable=true" - "traefik.enable=true"
- "traefik.http.middlewares.ipallowlist.ipallowlist.sourcerange=$WHITELIST" - "traefik.http.middlewares.ipallowlist.ipallowlist.sourcerange=${WHITELIST:-0.0.0.0/0}"
- "prometheus-scrape.enabled=true" - "prometheus-scrape.enabled=true"
- "prometheus-scrape.port=8082" - "prometheus-scrape.port=8082"
- "prometheus-scrape.job_name=traefik" - "prometheus-scrape.job_name=traefik"

View File

@@ -22,25 +22,26 @@ check_sync_status() {
# Cap the whole per-node branch (belt-and-suspenders over check-health's own cap), so no single # Cap the whole per-node branch (belt-and-suspenders over check-health's own cap), so no single
# node can ever block the 'wait' below — that is what wedged the fleet rpc-update for hours. # node can ever block the 'wait' below — that is what wedged the fleet rpc-update for hours.
result=$(timeout "${SYNC_TIMEOUT:-60}" "$BASEPATH/sync-status.sh" "${part%.yml}") result=$(timeout "${SYNC_TIMEOUT:-60}" "$BASEPATH/sync-status.sh" "${part%.yml}")
# Capture the status IMMEDIATELY. Any command in between - including a plain
# assignment like `code=0` - overwrites $? with its own (always 0) status.
rc=$?
code=0 code=0
if [ "$rc" -ne 0 ]; then
if [[ "$result" == *"syncing"* ]] || [[ "$result" == *"lagging"* ]]; then if [ $? -ne 0 ]; then
# sync-status exits 1 for syncing/lagging; those are expected states, if [[ "$result" == *"syncing"* ]]; then
# not failures. # Allow exit status 1 if result contains "syncing"
code=0
elif [[ "$result" == *"lagging"* ]]; then
# Allow exit status 1 if result contains "lagging"
code=0 code=0
else else
any_failure=true
code=1 code=1
fi fi
else
code=1
any_failure=true
fi fi
echo "${part%.yml}: $result" echo "${part%.yml}: $result"
# NOTE: do NOT set any_failure here. This function runs backgrounded (`&`), so
# it executes in a subshell and any variable it sets is discarded. Failure is
# propagated to the parent through this return code, collected by `wait` below.
return "$code" return "$code"
} }
@@ -73,12 +74,9 @@ for part in "${parts[@]}"; do
fi fi
done done
# Wait for all background processes to finish. `wait` runs in the PARENT shell, so # Wait for all background processes to finish
# this is where a failing node can actually flip any_failure - the checker itself
# cannot, being a subshell. Previously the status was discarded here, which silently
# neutered the exit code.
for pid in "${pids[@]}"; do for pid in "${pids[@]}"; do
wait "$pid" || any_failure=true wait "$pid"
done done
# Fenced nodes (fleet-state maintenance windows) are dropped from COMPOSE_FILE # Fenced nodes (fleet-state maintenance windows) are dropped from COMPOSE_FILE