From 0f522f2d8478e0db65a7b8f8b4d0ba5339a8fe58 Mon Sep 17 00:00:00 2001 From: Ayush Tiwari Date: Wed, 30 Sep 2026 07:33:45 +0530 Subject: [PATCH v2] Fix shutdown during crash restart A smart or fast shutdown during early crash restart can wait indefinitely for the new checkpointer and I/O workers. FatalError is still set, so the postmaster includes them in PM_WAIT_BACKENDS, but they ignore the SIGTERM sent by shutdown. AbortStartTime has already been reset as well. Use HandleFatalError(PMQUIT_FOR_STOP, false) to send SIGQUIT to the current children and arm the termination timeout. Keep FatalError and the existing startup-failure handling intact, limiting the change to shutdown rather than changing crash-restart behavior. Add a regression test that holds startup before WAL redo and waits for the restarted checkpointer to install its signal handlers before stopping. --- src/backend/postmaster/postmaster.c | 9 +++- src/test/recovery/meson.build | 1 + .../recovery/t/058_shutdown_crash_restart.pl | 54 +++++++++++++++++++ src/test/recovery/t/wait_for_shutdown | 19 +++++++ 4 files changed, 81 insertions(+), 2 deletions(-) create mode 100644 src/test/recovery/t/058_shutdown_crash_restart.pl create mode 100644 src/test/recovery/t/wait_for_shutdown diff --git a/src/backend/postmaster/postmaster.c b/src/backend/postmaster/postmaster.c index ef300a6c45a..af60e8b69ed 100644 --- a/src/backend/postmaster/postmaster.c +++ b/src/backend/postmaster/postmaster.c @@ -3041,9 +3041,14 @@ PostmasterStateMachine(void) */ ForgetUnstartedBackgroundWorkers(); - SignalChildren(SIGTERM, targetMask); + if (FatalError) + HandleFatalError(PMQUIT_FOR_STOP, false); + else + { + SignalChildren(SIGTERM, targetMask); - UpdatePMState(PM_WAIT_BACKENDS); + UpdatePMState(PM_WAIT_BACKENDS); + } } /* Are any of the target processes still running? */ diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build index ebb12dd8766..aee97e4da2a 100644 --- a/src/test/recovery/meson.build +++ b/src/test/recovery/meson.build @@ -66,6 +66,7 @@ tests += { 't/055_cascade_reconnect.pl', 't/056_standby_snapshot_export.pl', 't/057_snapshot_commit_race.pl', + 't/058_shutdown_crash_restart.pl', ], }, } diff --git a/src/test/recovery/t/058_shutdown_crash_restart.pl b/src/test/recovery/t/058_shutdown_crash_restart.pl new file mode 100644 index 00000000000..597806cf63a --- /dev/null +++ b/src/test/recovery/t/058_shutdown_crash_restart.pl @@ -0,0 +1,54 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Test fast shutdown during crash restart, before WAL redo has started. + +use strict; +use warnings FATAL => 'all'; +use FindBin; +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +my $node = PostgreSQL::Test::Cluster->new('node'); +$node->init(allows_streaming => 1); + +# Make the restarted startup process wait in restore_command until shutdown. +my $perlbin = $^X; +$perlbin =~ s!\\!/!g if $windows_os; +my $logfile = $node->logfile; +$logfile =~ s!\\!/!g if $windows_os; +my $timeout = $PostgreSQL::Test::Utils::timeout_default; +my $restore_timeout = 4 * $timeout; +$node->append_conf( + 'postgresql.conf', qq{ +restart_after_crash = on +log_min_messages = debug2 +restore_command = '"$perlbin" "$FindBin::RealBin/wait_for_shutdown" "$logfile" $restore_timeout' +}); +$node->start; + +$node->poll_query_until( + 'postgres', + q{SELECT count(*) = 1 FROM pg_stat_activity + WHERE backend_type = 'background writer'} +) or die 'background writer did not start'; +my $pid = $node->safe_psql('postgres', + "SELECT pid FROM pg_stat_activity WHERE backend_type = 'background writer'" +); +$node->set_standby_mode; +my $log_offset = -s $node->logfile; +system_or_bail('pg_ctl', 'kill', 'QUIT', $pid); +$node->wait_for_log(qr/restore_command waiting for shutdown/, $log_offset); +# Wait until the new checkpointer has installed its SIGTERM handler. +$node->wait_for_log( + qr/checkpointer updated shared memory configuration values/, $log_offset); + +ok( $node->stop('fast', fail_ok => 1, timeout => $timeout), + 'fast shutdown completes during crash restart'); +# pg_ctl can report success after a helper timeout made startup fail. +unlike( + slurp_file($node->logfile, $log_offset), + qr/timed out waiting for shutdown request/, + 'restore_command did not time out'); + +done_testing(); diff --git a/src/test/recovery/t/wait_for_shutdown b/src/test/recovery/t/wait_for_shutdown new file mode 100644 index 00000000000..8c9ceb5b70a --- /dev/null +++ b/src/test/recovery/t/wait_for_shutdown @@ -0,0 +1,19 @@ +#!/usr/bin/perl + +# restore_command helper: wait until the server log shows a shutdown request. + +use strict; +use warnings FATAL => 'all'; +use Time::HiRes qw(usleep); + +my ($logfile, $timeout) = @ARGV; + +print STDERR "restore_command waiting for shutdown\n"; +for (1 .. $timeout * 10) +{ + open my $fh, '<', $logfile or die "could not open $logfile: $!"; + exit 1 if grep { /received \w+ shutdown request/ } <$fh>; + close $fh; + usleep(100_000); +} +die "timed out waiting for shutdown request\n"; base-commit: 6a93535798aa9219d01c01c67161a78780e2af1a -- 2.34.1