#!/bin/sh
# A fast shutdown issued right after a backend crash, while the restarted
# startup process is still syncing the data directory (before redo), never
# completes: the postmaster keeps waiting for the checkpointer (and, from 18,
# the IO workers), which ignore the SIGTERM they are sent.
# Many relation files make the startup process's sync long enough to hit.
#
# usage: repro_shutdown_hang.sh <bindir> [port]
B=$1; PORT=${2:-5499}
D=$(mktemp -d)/data
"$B/initdb" -D "$D" -A trust >/dev/null || exit 1
echo "port = $PORT" >> "$D/postgresql.conf"
"$B/pg_ctl" -D "$D" -l "$D/log" -w start >/dev/null || exit 1
"$B/psql" -X -q -p "$PORT" -d postgres \
  -c "DO \$\$ BEGIN FOR i IN 1..4000 LOOP EXECUTE format('CREATE TABLE t%s (a int PRIMARY KEY)', i); IF i % 500 = 0 THEN COMMIT; END IF; END LOOP; END \$\$" \
  -c "CHECKPOINT" || exit 1
PM=$(head -1 "$D/postmaster.pid")
"$B/psql" -X -p "$PORT" -d postgres -c "SELECT pg_sleep(60)" >/dev/null 2>&1 &
sleep 1
BP=$("$B/psql" -X -At -p "$PORT" -d postgres -c "SELECT pid FROM pg_stat_activity WHERE query = 'SELECT pg_sleep(60)'")
kill -SEGV "$BP"
until grep -q "reinitializing" "$D/log"; do sleep 0.01; done
if "$B/pg_ctl" -D "$D" -m fast -t 60 stop; then
  echo "fast shutdown completed"
else
  echo "fast shutdown did not complete; the postmaster's children:"
  ps --ppid "$PM" -o pid=,cmd=
  "$B/pg_ctl" -D "$D" -m immediate stop
fi
grep -E "reinitializing|fast shutdown request|automatic recovery|redo starts" "$D/log"
rm -rf "$(dirname "$D")"
