From 47cd2a5ff79d61cad73c3a7732c3d6386e5db2ea Mon Sep 17 00:00:00 2001 From: Michael Paquier Date: Thu, 1 Oct 2026 13:16:50 +0900 Subject: [PATCH v3] Fix handling of shutdown requests during a early crash restart A smart or fast shutdown during early crash restart can wait indefinitely for the new checkpointer and I/O workers. FatalError is still set, so the postmaster includes them in PM_WAIT_BACKENDS, but they ignore the SIGTERM sent by the shutdown. If the postmaster is requesting for the backends to stop while FatalError is still one, this upgrades the signal sent to the backends to SIGQUIT, sent by HandleFatalError(), instead of SIGTERM. A regression test is added, that relies on a restore_command waiting for a shutdown request, keeping the startup process at some very early stage of recovery. No backpatch is done for now, to be conservative. Reported-by: Justin Pryzby Discussion: https://postgr.es/m/ --- src/backend/postmaster/postmaster.c | 14 ++++- src/test/recovery/meson.build | 1 + .../recovery/t/058_shutdown_crash_restart.pl | 61 +++++++++++++++++++ src/test/recovery/t/wait_for_shutdown | 22 +++++++ 4 files changed, 96 insertions(+), 2 deletions(-) create mode 100644 src/test/recovery/t/058_shutdown_crash_restart.pl create mode 100644 src/test/recovery/t/wait_for_shutdown diff --git a/src/backend/postmaster/postmaster.c b/src/backend/postmaster/postmaster.c index ef300a6c45a6..f53aa9c404d5 100644 --- a/src/backend/postmaster/postmaster.c +++ b/src/backend/postmaster/postmaster.c @@ -3041,9 +3041,19 @@ PostmasterStateMachine(void) */ ForgetUnstartedBackgroundWorkers(); - SignalChildren(SIGTERM, targetMask); + /* + * While processing a crash, targetMask includes the checkpointer + * and the io workers. These ignore SIGTERM, so upgrade to + * SIGQUIT. + */ + if (FatalError) + HandleFatalError(PMQUIT_FOR_STOP, false); + else + { + SignalChildren(SIGTERM, targetMask); - UpdatePMState(PM_WAIT_BACKENDS); + UpdatePMState(PM_WAIT_BACKENDS); + } } /* Are any of the target processes still running? */ diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build index ebb12dd87665..aee97e4da2a0 100644 --- a/src/test/recovery/meson.build +++ b/src/test/recovery/meson.build @@ -66,6 +66,7 @@ tests += { 't/055_cascade_reconnect.pl', 't/056_standby_snapshot_export.pl', 't/057_snapshot_commit_race.pl', + 't/058_shutdown_crash_restart.pl', ], }, } diff --git a/src/test/recovery/t/058_shutdown_crash_restart.pl b/src/test/recovery/t/058_shutdown_crash_restart.pl new file mode 100644 index 000000000000..e46b3fe04cf2 --- /dev/null +++ b/src/test/recovery/t/058_shutdown_crash_restart.pl @@ -0,0 +1,61 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Test shutdown during a crash restart, before WAL redo has started. The +# test relies on a fake restore_command that keeps the startup process at +# some early stage, waiting for a shutdown to happen, with a checkpointer +# spawned and running. + +use strict; +use warnings FATAL => 'all'; +use FindBin; +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +my $node = PostgreSQL::Test::Cluster->new('node'); +$node->init(allows_streaming => 1); + +# Make the restarted startup process wait in restore_command until shutdown. +my $perlbin = $^X; +$perlbin =~ s!\\!/!g if $windows_os; +my $logfile = $node->logfile; +$logfile =~ s!\\!/!g if $windows_os; +my $restore_timeout = $PostgreSQL::Test::Utils::timeout_default; + +# DEBUG2 is required for the checkpointer log entry lookup. +$node->append_conf( + 'postgresql.conf', qq{ +restart_after_crash = on +log_min_messages = debug2 +restore_command = '"$perlbin" "$FindBin::RealBin/wait_for_shutdown" "$logfile" $restore_timeout' +}); +$node->start; + +# Stop the background writer. +$node->poll_query_until( + 'postgres', + q{SELECT count(*) = 1 FROM pg_stat_activity + WHERE backend_type = 'background writer'} +) or die 'background writer did not start'; +my $pid = $node->safe_psql('postgres', + "SELECT pid FROM pg_stat_activity WHERE backend_type = 'background writer'" +); +$node->set_standby_mode; +my $log_offset = -s $node->logfile; +is(PostgreSQL::Test::Utils::system_log('pg_ctl', 'kill', 'QUIT', $pid), + 0, "SIGQUIT sent to background writer"); +$node->wait_for_log(qr/restore_command waiting for shutdown/, $log_offset); + +# Wait until the new checkpointer has installed its SIGTERM handler. +$node->wait_for_log( + qr/checkpointer updated shared memory configuration values/, $log_offset); + +ok($node->stop('fast', fail_ok => 1), + 'fast shutdown completes during crash restart'); + +unlike( + slurp_file($node->logfile, $log_offset), + qr/timed out waiting for shutdown request/, + 'restore_command did not time out'); + +done_testing(); diff --git a/src/test/recovery/t/wait_for_shutdown b/src/test/recovery/t/wait_for_shutdown new file mode 100644 index 000000000000..2d121f692676 --- /dev/null +++ b/src/test/recovery/t/wait_for_shutdown @@ -0,0 +1,22 @@ +#!/usr/bin/perl + +# restore_command helper: wait until the server log shows a shutdown request. +# This script accepts two arguments: +# - A log file to monitor, inherited from the server spawned. +# - A timeout value, defined by $PostgreSQL::Test::Utils::timeout_default. + +use strict; +use warnings FATAL => 'all'; +use Time::HiRes qw(usleep); + +my ($logfile, $timeout) = @ARGV; + +print STDERR "restore_command waiting for shutdown\n"; +for (my $i = 0; $i < $timeout * 10; $i++) +{ + open my $fh, '<', $logfile or die "could not open $logfile: $!"; + exit 1 if grep { /received \w+ shutdown request/ } <$fh>; + close $fh; + usleep(100_000); +} +die "timed out waiting for shutdown request\n"; -- 2.55.0