Files
ethereum-rpc-docker/reload_dshackle.sh
rob 6f4b297733 reload_dshackle: cut gateway connections as soon as the SIGHUP handler reports completion (~30ms) instead of after a fixed 3s settle
The reload is in-process and takes 10-30 ms; the fixed sleep was the window in
which a removed upstream was already unrouted in dshackle while the gateway still
held the old announcements and kept sending queries it could no longer answer.
Poll the log for 'Reloading config has been completed' (or the fail/drop
markers), SETTLE stays the upper bound. Operator 2026-09-19.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01KjqejYycVyVvrxV9aTDhgX
2026-09-19 09:50:03 +00:00

272 lines
11 KiB
Bash
Executable File

#!/bin/bash
# Re-weave host-local upstream overrides (method disables etc.) into the freshly
# pushed configs BEFORE signaling dshackle - makes /root/rpc-local/dshackle-overrides.yaml
# survive every configure-drpc regeneration. See apply-dshackle-overrides.py.
[ -f /root/rpc/apply-dshackle-overrides.py ] && python3 /root/rpc/apply-dshackle-overrides.py
# --- Snapshot-based conditional gateway reconnect ---
# Operator ruling (2026-08-27): NEVER force a reconnect on a routine reload/rpc-update run.
# Only upstream REMOVALS or new disabled-method ADDITIONS require a gateway reconnect.
#
# For each dshackle container, we snapshot the advertised set:
# (a) upstream ids present in rendered yaml configs
# (b) disabled-method entries (from methods.disabled across all upstreams)
# into a persisted file: /root/rpc/.dshackle-adv-<container>.snapshot
#
# On each run:
# - Read current config state from disk
# - If previous snapshot exists, compare: drop connections ONLY IF
# * an upstream id disappeared (was in snapshot, not in current), OR
# * a new disabled-method entry appeared (in current, not in snapshot)
# - Pure additions or no change: reload only, no drop
# - First run with no prior snapshot: do NOT drop, just write snapshot after reload
# - Always rewrite snapshot after successful reload (for next run to compare against)
# - All snapshot/compare operations are best-effort; failures never fail the script
# Helper: extract the host config directory path for a dshackle container.
# dshackle containers mount their config at /etc/dshackle; we find the source of that mount.
get_config_dir() {
local cid="$1"
local mount
mount=$(docker inspect -f '{{range .Mounts}}{{if eq .Destination "/etc/dshackle"}}{{.Source}}{{end}}{{end}}' "$cid" 2>/dev/null) || true
[ -n "$mount" ] && echo "$mount" && return 0
# Fallback: try common locations (best-effort)
for d in /root/rpc/main_configs /root/rpc/free_configs; do
[ -d "$d" ] && echo "$d" && return 0
done
return 1
}
# Helper: extract advertised set (upstream ids + disabled methods) from config directory.
# Outputs two lines to stdout:
# upstream_ids (space-separated, sorted)
# disabled_methods (space-separated, sorted)
get_advertised_set() {
local config_dir="$1"
local upstream_ids=""
local disabled_methods=""
[ -d "$config_dir" ] || return 1
# Use python3 to parse YAML (available on host). Best-effort: if python fails, return empty.
python3 -c "
import yaml, glob, sys, os
config_dir = sys.argv[1]
upstream_ids = set()
disabled_methods = set()
for path in glob.glob(os.path.join(config_dir, '*.yaml')):
try:
with open(path) as f:
doc = yaml.safe_load(f)
if not isinstance(doc, dict):
continue
for up in doc.get('upstreams') or []:
if not isinstance(up, dict):
continue
uid = up.get('id')
if uid:
upstream_ids.add(uid)
methods = up.get('methods', {})
if isinstance(methods, dict):
for m in methods.get('disabled') or []:
if isinstance(m, dict):
name = m.get('name')
if name:
disabled_methods.add(name)
elif isinstance(m, str):
disabled_methods.add(m)
except Exception:
pass
print(' '.join(sorted(upstream_ids)))
print(' '.join(sorted(disabled_methods)))
" "$config_dir" 2>/dev/null
return 0
}
# Helper: read snapshot file. Returns upstream_ids and disabled_methods.
read_snapshot() {
local snapshot_file="$1"
[ -f "$snapshot_file" ] || return 1
local line1 line2
line1=$(head -1 "$snapshot_file" 2>/dev/null) || true
line2=$(tail -1 "$snapshot_file" 2>/dev/null) || true
[ -n "$line1" ] && echo "$line1"
[ -n "$line2" ] && echo "$line2"
return 0
}
# Helper: write snapshot file.
write_snapshot() {
local snapshot_file="$1"
local upstream_ids="$2"
local disabled_methods="$3"
echo "$upstream_ids" > "$snapshot_file" 2>/dev/null || true
echo "$disabled_methods" >> "$snapshot_file" 2>/dev/null || true
}
# Helper: check if drop is needed.
# Args: old_upstream_ids, old_disabled_methods, new_upstream_ids, new_disabled_methods
# Returns 0 (true) if drop is needed, 1 (false) otherwise.
drop_needed() {
local old_ids="$1" new_ids="$3"
local old_methods="$2" new_methods="$4"
# Check if any upstream id disappeared (in old, not in new)
for uid in $old_ids; do
[ -z "$uid" ] && continue
# Check if uid is NOT in new_ids
found=0
for nuid in $new_ids; do
[ "$uid" = "$nuid" ] && found=1 && break
done
[ "$found" = 0 ] && return 0 # drop needed
done
# Check if any new disabled method appeared (in new, not in old)
for meth in $new_methods; do
[ -z "$meth" ] && continue
found=0
for ometh in $old_methods; do
[ "$meth" = "$ometh" ] && found=1 && break
done
[ "$found" = 0 ] && return 0 # drop needed
done
return 1 # no drop needed
}
# 1:1 invariant (2026-07-15): refuse to ACTIVATE a dshackle config routing >1 node for the same
# chain — we can't attribute traffic to multiple nodes behind one proxy (no per-upstream request
# metric; conn-seconds biases it), and the attribution model + planner assume 1:1. The validator
# reads /root/rpc/main_configs/*.yaml; on violation it exits 1 and we keep the live config.
# See /root/proxy-1to1-invariant-plan.md + rpc/validate-dshackle-1to1.py.
if [ -f /root/rpc/validate-dshackle-1to1.py ]; then
python3 /root/rpc/validate-dshackle-1to1.py /root/rpc/main_configs || {
echo "reload_dshackle.sh: REFUSING reload — 1:1 invariant violated (above); keeping live config" >&2
exit 1
}
fi
# Signal dshackle to reload, then VERIFY the reload applied. dshackle's SIGHUP reload is
# unreliable for upstream REMOVALS (and method-set changes, same remove+add path): it can
# throw internally, log "Config is not reloaded, cause - ...", and SILENTLY keep serving the
# old config (in-memory config updated, runtime selectors stale = drift). Note
# "Reloading config has been completed" is NOT success — it's a finally-block and prints even
# when a processor threw. There is no runtime admin API to mutate upstreams, so a container
# restart is the only way to actually apply such changes.
# Per container: SIGHUP, check the post-SIGHUP logs for a failure/drop marker, restart on
# failure. Exit 0 when the config ends up applied (via SIGHUP or restart); non-zero only if a
# needed restart failed or the container didn't come back (so the deploy fails loudly instead
# of silently drifting).
set -u
FAIL_MARK='Config is not reloaded, cause -' # a reload processor threw (removal/method bug)
DROP_MARK='Reloading is in progress' # a concurrent HUP was dropped (reload skipped)
SETTLE=3 # seconds for the SIGHUP handler to run + log
RC=0
for CID in $(docker ps -q -f "name=dshackle"); do
NAME=$(docker inspect -f '{{.Name}}' "$CID" 2>/dev/null | sed 's|^/||')
[ -n "$NAME" ] || NAME="$CID"
# --- Snapshot-based conditional gateway reconnect ---
# Determine config directory and snapshot file for this container.
CONFIG_DIR=$(get_config_dir "$CID") || true
# Sanitize name for use in snapshot filename: replace / with -, remove leading -
SANITIZED_NAME=$(echo "$NAME" | tr '/' '-' | sed 's/^-//')
SNAPSHOT_FILE="/root/rpc/.dshackle-adv-${SANITIZED_NAME}.snapshot"
# Read current advertised set from the (new) on-disk config.
CURRENT_SET=$(get_advertised_set "$CONFIG_DIR") || true
CURRENT_UPSTREAMS=$(echo "$CURRENT_SET" | head -1 | sed 's/^[[:space:]]*//;s/[[:space:]]*$//')
CURRENT_METHODS=$(echo "$CURRENT_SET" | tail -1 | sed 's/^[[:space:]]*//;s/[[:space:]]*$//')
# Read previous snapshot if it exists.
PREVIOUS_UPSTREAMS=""
PREVIOUS_METHODS=""
HAS_PREVIOUS_SNAPSHOT=0
if [ -f "$SNAPSHOT_FILE" ]; then
PREVIOUS_SET=$(read_snapshot "$SNAPSHOT_FILE") || true
PREVIOUS_UPSTREAMS=$(echo "$PREVIOUS_SET" | head -1 | sed 's/^[[:space:]]*//;s/[[:space:]]*$//')
PREVIOUS_METHODS=$(echo "$PREVIOUS_SET" | tail -1 | sed 's/^[[:space:]]*//;s/[[:space:]]*$//')
HAS_PREVIOUS_SNAPSHOT=1
fi
# Decide if we need to drop connections: only if there were removals.
NEED_DROP=0
if [ "$HAS_PREVIOUS_SNAPSHOT" -eq 1 ]; then
# Has previous snapshot: check for removals
if drop_needed "$PREVIOUS_UPSTREAMS" "$PREVIOUS_METHODS" "$CURRENT_UPSTREAMS" "$CURRENT_METHODS"; then
NEED_DROP=1
fi
fi
# If no previous snapshot, NEED_DROP stays 0 (first run: do NOT drop)
T0=$(date +%s)
if ! docker kill --signal=HUP "$CID" >/dev/null 2>&1; then
echo "reload_dshackle: WARNING: failed to SIGHUP $NAME" >&2
RC=1
continue
fi
# Poll for the handler's completion line instead of sleeping SETTLE: the reload itself
# takes ~10-30 ms, but the fixed 3 s sleep was the window in which a removed upstream
# was already gone while the gateway still held the old announcements (operator
# 2026-09-19). SETTLE is now the upper bound only.
DONE_MARK="Reloading config has been completed"
LOGS=""
for _ in $(seq 1 $(( SETTLE * 20 ))); do
LOGS=$(docker logs --since "$T0" "$CID" 2>&1)
if echo "$LOGS" | grep -qF "$DONE_MARK" || echo "$LOGS" | grep -qF "$FAIL_MARK" || echo "$LOGS" | grep -qF "$DROP_MARK"; then
break
fi
sleep 0.05
done
REASON=""
if echo "$LOGS" | grep -qF "$FAIL_MARK"; then
REASON=$(echo "$LOGS" | grep -F "$FAIL_MARK" | head -1)
elif echo "$LOGS" | grep -qF "$DROP_MARK"; then
REASON="concurrent HUP dropped (reload skipped)"
fi
[ -z "$REASON" ] && {
# Reload succeeded. Always rewrite snapshot for next run.
write_snapshot "$SNAPSHOT_FILE" "$CURRENT_UPSTREAMS" "$CURRENT_METHODS"
# Drop established gateway gRPC connections ONLY if removals were detected.
if [ "$NEED_DROP" -eq 1 ]; then
PID=$(docker inspect -f '{{.State.Pid}}' "$CID" 2>/dev/null) && \
nsenter -t "$PID" -n ss -K state established '( sport = :2449 )' >/dev/null 2>&1
echo "reload_dshackle: $NAME reload applied, gateway connections dropped (removals detected)" >&2
else
echo "reload_dshackle: $NAME reload applied, no removals — gateway connections kept" >&2
fi
continue
}
echo "reload_dshackle: $NAME reload did NOT apply ($REASON) — restarting to load config fresh" >&2
if ! docker restart "$CID" >/dev/null 2>&1; then
echo "reload_dshackle: ERROR: restart failed for $NAME — config NOT applied, manual intervention" >&2
RC=1
continue
fi
UP=""
for _ in $(seq 1 30); do
if docker inspect -f '{{.State.Running}}' "$CID" 2>/dev/null | grep -q true; then UP=1; break; fi
sleep 2
done
if [ -n "$UP" ]; then
# Restart succeeded. Always rewrite snapshot for next run.
write_snapshot "$SNAPSHOT_FILE" "$CURRENT_UPSTREAMS" "$CURRENT_METHODS"
echo "reload_dshackle: $NAME restarted, config applied on clean start" >&2
# On restart, the config is freshly loaded, so we need to drop if removals were detected.
if [ "$NEED_DROP" -eq 1 ]; then
PID=$(docker inspect -f '{{.State.Pid}}' "$CID" 2>/dev/null) && \
nsenter -t "$PID" -n ss -K state established '( sport = :2449 )' >/dev/null 2>&1
echo "reload_dshackle: $NAME restart applied, gateway connections dropped (removals detected)" >&2
fi
else
echo "reload_dshackle: ERROR: $NAME not running after restart — manual intervention" >&2
RC=1
fi
done
exit $RC