Add db-migrations-expand-contract: zero-downtime schema migrations proven with a real 4-deploy rolling run
Companion code for Zero-Downtime Database Migrations: Expand-Contract in Practice with Spring Boot: a full expand/migrate-writes/migrate-reads/contract sequence run as an actual rolling deploy across two live replicas, with a load generator sending continuous HTTP traffic through all four deploys (99.98% success, every residual error traced to a root cause rather than left unexplained). Findings include a real NOT NULL constraint trap in the expand migration, a backfill-window bug in the read switch, H2's AUTO_SERVER=TRUE single-point-of-failure behavior under a rolling restart, the drain-before-SIGTERM fix needed to close a health-check gap during graceful shutdown, and H2 silently discarding a concurrently committed INSERT during an ALTER TABLE ADD/DROP COLUMN rebuild - confirmed, by primary source, to be an H2-specific behavior rather than a property of the technique itself. Co-Authored-By: Claude Sonnet 5 <[email protected]> Claude-Session: https://claude.ai/code/session_019Fb7vW8vLyLKngBc4R3huA
This commit is contained in:
+145
@@ -0,0 +1,145 @@
|
||||
#!/usr/bin/env bash
|
||||
# The live exhibit: two replicas, a shared H2 database, and a load generator that
|
||||
# never stops sending traffic while this script performs all four deploys of the
|
||||
# expand-contract sequence as an actual rolling deploy - one replica at a time.
|
||||
#
|
||||
# Regenerates:
|
||||
# docs/output/11-live-deploy-sequence.txt - the deploy log, phase by phase
|
||||
# docs/output/12-load-generator-summary.txt - total/ok/error counts, by phase
|
||||
# docs/output/13-schema-diagnostics-timeline.txt - /diag/schema after each deploy
|
||||
set -euo pipefail
|
||||
DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
source "$DIR/env.sh"
|
||||
|
||||
OUTPUT_DIR="$MODULE_DIR/docs/output"
|
||||
mkdir -p "$OUTPUT_DIR"
|
||||
DEPLOY_LOG="$OUTPUT_DIR/11-live-deploy-sequence.txt"
|
||||
DIAG_LOG="$OUTPUT_DIR/13-schema-diagnostics-timeline.txt"
|
||||
|
||||
log() {
|
||||
echo "$(date -u +%H:%M:%S) $*" | tee -a "$DEPLOY_LOG"
|
||||
}
|
||||
|
||||
diag() {
|
||||
local label="$1"
|
||||
{
|
||||
echo ""
|
||||
echo "-- $label --"
|
||||
curl -s "http://localhost:$PORT_A/diag/schema"
|
||||
echo ""
|
||||
} | tee -a "$DIAG_LOG"
|
||||
}
|
||||
|
||||
# --- clean slate ---
|
||||
for p in "$PORT_A" "$PORT_B"; do
|
||||
"$DIR/stop-instance.sh" "$p" 2>/dev/null || true
|
||||
done
|
||||
if [ -f "$EC_PID_DIR/loadgen.pid" ]; then
|
||||
kill -9 "$(cat "$EC_PID_DIR/loadgen.pid")" 2>/dev/null || true
|
||||
rm -f "$EC_PID_DIR/loadgen.pid"
|
||||
fi
|
||||
"$DIR/stop-db-server.sh" 2>/dev/null || true
|
||||
rm -rf "$EC_DB_BASE_DIR"
|
||||
mkdir -p "$EC_DB_BASE_DIR"
|
||||
rm -f "$DEPLOY_LOG" "$DIAG_LOG"
|
||||
|
||||
echo "=====================================================================" > "$DEPLOY_LOG"
|
||||
echo "Zero-downtime expand-contract: live 4-deploy sequence" >> "$DEPLOY_LOG"
|
||||
echo "=====================================================================" >> "$DEPLOY_LOG"
|
||||
log "captured: $(date -u +%FT%TZ)"
|
||||
|
||||
# --- Deploy -1: the database itself. This process is never restarted for the rest
|
||||
# of this script - it is the one thing every deploy below has to treat as always up.
|
||||
log "Starting the database as its own standalone process (not owned by either replica)"
|
||||
"$DIR/start-db-server.sh" | tee -a "$DEPLOY_LOG"
|
||||
|
||||
# --- Deploy 0: baseline schema + two Stage 1 replicas ---
|
||||
log "Deploy 0: create schema (V1), start two Stage 1 replicas"
|
||||
"$DIR/migrate.sh" 1 | tee -a "$DEPLOY_LOG" > /dev/null
|
||||
"$DIR/start-instance.sh" "$PORT_A" 1 | tee -a "$DEPLOY_LOG"
|
||||
"$DIR/start-instance.sh" "$PORT_B" 1 | tee -a "$DEPLOY_LOG"
|
||||
diag "after Deploy 0 (Stage 1 / Stage 1)"
|
||||
|
||||
"$DIR/run-load-generator.sh" 120
|
||||
log "load generator running against both replicas"
|
||||
sleep 10
|
||||
log "baseline soak complete (10s, both replicas on Stage 1)"
|
||||
|
||||
# --- Deploy 1 (EXPAND): schema only, zero app restarts ---
|
||||
set_phase "01-expand-migration"
|
||||
log "Deploy 1 (EXPAND): migrating to V2 live - zero app restarts"
|
||||
"$DIR/migrate.sh" 2 | tee -a "$DEPLOY_LOG" > /dev/null
|
||||
diag "after Deploy 1 (schema expanded, both replicas still Stage 1)"
|
||||
sleep 5
|
||||
|
||||
# --- Deploy 2 (MIGRATE WRITES): rolling restart to Stage 2 ---
|
||||
set_phase "02-deploy-stage2-rollout"
|
||||
log "Deploy 2 (MIGRATE WRITES): rolling restart to Stage 2, replica A first"
|
||||
"$DIR/stop-instance.sh" "$PORT_A" | tee -a "$DEPLOY_LOG"
|
||||
"$DIR/start-instance.sh" "$PORT_A" 2 | tee -a "$DEPLOY_LOG"
|
||||
sleep 2
|
||||
log "Deploy 2: replica B"
|
||||
"$DIR/stop-instance.sh" "$PORT_B" | tee -a "$DEPLOY_LOG"
|
||||
"$DIR/start-instance.sh" "$PORT_B" 2 | tee -a "$DEPLOY_LOG"
|
||||
set_phase "02-stage2-soak"
|
||||
diag "after Deploy 2 (both replicas Stage 2, dual-write live)"
|
||||
sleep 8
|
||||
|
||||
# --- Deploy 3 (MIGRATE READS): rolling restart to Stage 3 ---
|
||||
set_phase "03-deploy-stage3-rollout"
|
||||
log "Deploy 3 (MIGRATE READS): rolling restart to Stage 3, replica A first"
|
||||
"$DIR/stop-instance.sh" "$PORT_A" | tee -a "$DEPLOY_LOG"
|
||||
"$DIR/start-instance.sh" "$PORT_A" 3 | tee -a "$DEPLOY_LOG"
|
||||
sleep 2
|
||||
log "Deploy 3: replica B"
|
||||
"$DIR/stop-instance.sh" "$PORT_B" | tee -a "$DEPLOY_LOG"
|
||||
"$DIR/start-instance.sh" "$PORT_B" 3 | tee -a "$DEPLOY_LOG"
|
||||
set_phase "03-stage3-soak"
|
||||
diag "after Deploy 3 (both replicas Stage 3, reading email_address)"
|
||||
sleep 8
|
||||
|
||||
# --- Deploy 4a (CONTRACT, code): rolling restart to Stage 4 ---
|
||||
set_phase "04a-deploy-stage4-rollout"
|
||||
log "Deploy 4a (CONTRACT code): rolling restart to Stage 4, replica A first"
|
||||
"$DIR/stop-instance.sh" "$PORT_A" | tee -a "$DEPLOY_LOG"
|
||||
"$DIR/start-instance.sh" "$PORT_A" 4 | tee -a "$DEPLOY_LOG"
|
||||
sleep 2
|
||||
log "Deploy 4a: replica B"
|
||||
"$DIR/stop-instance.sh" "$PORT_B" | tee -a "$DEPLOY_LOG"
|
||||
"$DIR/start-instance.sh" "$PORT_B" 4 | tee -a "$DEPLOY_LOG"
|
||||
set_phase "04a-stage4-soak"
|
||||
diag "after Deploy 4a (both replicas Stage 4, email column still present but unused)"
|
||||
sleep 8
|
||||
|
||||
# --- Deploy 4b (CONTRACT, schema): drop the old column, live ---
|
||||
set_phase "04b-contract-migration"
|
||||
log "Deploy 4b (CONTRACT schema): migrating to V3 live - drops \"email\", zero app restarts"
|
||||
"$DIR/migrate.sh" latest | tee -a "$DEPLOY_LOG" > /dev/null
|
||||
diag "after Deploy 4b (email column dropped)"
|
||||
|
||||
set_phase "05-final-soak"
|
||||
sleep 10
|
||||
log "final soak complete"
|
||||
|
||||
log "waiting for the load generator to finish its run..."
|
||||
if [ -f "$EC_PID_DIR/loadgen.pid" ]; then
|
||||
LOADGEN_PID="$(cat "$EC_PID_DIR/loadgen.pid")"
|
||||
while kill -0 "$LOADGEN_PID" 2>/dev/null; do
|
||||
sleep 1
|
||||
done
|
||||
fi
|
||||
log "load generator finished"
|
||||
|
||||
cp "$EC_LOAD_SUMMARY" "$OUTPUT_DIR/12-load-generator-summary.txt"
|
||||
|
||||
for p in "$PORT_A" "$PORT_B"; do
|
||||
"$DIR/stop-instance.sh" "$p" || true
|
||||
done
|
||||
"$DIR/stop-db-server.sh" || true
|
||||
|
||||
log "=== deploy sequence complete ==="
|
||||
echo ""
|
||||
echo "Deploy log: $DEPLOY_LOG"
|
||||
echo "Load summary: $OUTPUT_DIR/12-load-generator-summary.txt"
|
||||
echo "Schema timeline: $DIAG_LOG"
|
||||
cat "$OUTPUT_DIR/12-load-generator-summary.txt"
|
||||
Reference in New Issue
Block a user