diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build index 7982bb6b0ba..20299e25f7d 100644 --- a/src/test/recovery/meson.build +++ b/src/test/recovery/meson.build @@ -67,6 +67,7 @@ tests += { 't/056_standby_snapshot_export.pl', 't/057_snapshot_commit_race.pl', 't/059_remote_apply_status_interval.pl', + 't/060_repack_postmaster_exit.pl', ], }, } diff --git a/src/test/recovery/t/060_repack_postmaster_exit.pl b/src/test/recovery/t/060_repack_postmaster_exit.pl new file mode 100644 index 00000000000..22ef3bd5f33 --- /dev/null +++ b/src/test/recovery/t/060_repack_postmaster_exit.pl @@ -0,0 +1,140 @@ +# Copyright (c) 2026 PostgreSQL Global Development Group +# +# Test that a backend running REPACK (CONCURRENTLY) exits cleanly if the +# postmaster is killed while the command is in progress. Killing the +# postmaster during the "rebuilding index" phase used to make the REPACK +# steering backend fail the !IsTransactionOrTransactionBlock() assertion +# in pgstat_report_stat() (or, in a non-cassert build, flush statistics +# from an inconsistent transaction state), because FATAL errors raised on +# the process-exit path re-enter proc_exit()/shmem_exit() while the +# AbortOutOfAnyTransaction() started by the ShutdownPostgres exit callback +# has not completed yet. Cover the same exit path for a plain parallel +# index build, which on postmaster death raises a FATAL error while +# waiting for its workers. + +use strict; +use warnings FATAL => 'all'; +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +my $node = PostgreSQL::Test::Cluster->new('primary'); +$node->init(); +$node->append_conf('postgresql.conf', 'wal_level = logical'); +$node->start(); + +# Build a table big enough that the interesting progress phases last a +# while. +$node->safe_psql( + 'postgres', q( + CREATE TABLE repacked (id serial PRIMARY KEY, val text, num int);)); +$node->safe_psql( + 'postgres', q( + INSERT INTO repacked(val, num) + SELECT repeat('v', 200)||g::text, g FROM generate_series(1, 800000) g;)); +$node->safe_psql('postgres', q(CREATE INDEX repacked_num ON repacked(num);)); + +run_kill_test( + 'REPACK', 'pg_stat_progress_repack', 'rebuilding index', + q(REPACK (CONCURRENTLY) repacked;)); + +run_kill_test( + 'parallel CREATE INDEX', 'pg_stat_progress_create_index', + 'building index: scanning table', + q(SET max_parallel_maintenance_workers = 4; + CREATE INDEX repacked_num2 ON repacked(num);)); + +done_testing(); + +sub run_kill_test +{ + my ($name, $progress_view, $target_phase, $command) = @_; + my $marker = "COMMAND_FINISHED"; + + # Only look at log lines written from now on. + my $log_offset = -s $node->logfile; + + # Run the command in the background. + my $session = $node->background_psql('postgres'); + $session->query_until(qr/started/, + "\\echo started\n$command\n\\echo $marker\n"); + + # Wait until the command reaches the wanted progress phase. + ok( $node->poll_query_until( + 'postgres', + qq[SELECT phase = '$target_phase' FROM $progress_view;]), + "$name: command reached phase '$target_phase'"); + + # Kill the postmaster, orphaning all its children. Remember the + # children, so that any orphan that does not exit on its own can be + # killed below; the node has to recover from all of them dying anyway. + my $pm_pid; + { + my $pidfile = $node->data_dir . "/postmaster.pid"; + open(my $fh, '<', $pidfile) or die "could not open $pidfile: $!"; + chomp($pm_pid = <$fh>); + close($fh); + } + my @child_pids = map { $_ + 0 } split /\n/, + $node->safe_psql('postgres', "SELECT pid FROM pg_stat_activity"); + + kill(9, $pm_pid); + for my $pid (@child_pids) + { + next if $pid == $pm_pid; + my $tries = 0; + while (kill(0, $pid) == 1 && $tries++ < 100) + { + select undef, undef, undef, 0.1; + } + kill(9, $pid) if kill(0, $pid) == 1; + } + + # Wait until the orphaned psql session is gone. There is no clean way + # to shut down psql after the postmaster is gone; if it did not exit on + # its own with the dead connection, sending \q might throw a "broken + # pipe", which can be ignored. + eval { + $session->quit(); + 1; + } or note("psql session exited: $@"); + + # Cluster.pm has no interface for killing a postmaster directly, so + # update its idea of the currently running postmaster ourselves, and + # remove the stale pid file. + $node->{_pid} = undef; + + # The orphaned children may keep the shared memory block in use for a + # short while after the postmaster is gone, so keep retrying the start + # until they are gone. + my $started = 0; + for (my $i = 0; $i < 30 and not $started; $i++) + { + unlink($node->data_dir . "/postmaster.pid"); + $started = $node->start(fail_ok => 1); + sleep(1) unless $started; + } + ok($started, "$name: node restarted after crash recovery"); + + # The exit path of the killed statement must have run cleanly: no + # assertion failure in the log (this catches the broken nested + # proc_exit()/shmem_exit() unwinding). + ok( !$node->log_contains(qr/TRAP: failed Assert/, $log_offset), + "$name: no assertion failure in the server log"); + + # The cluster must be consistent and the same kind of command must + # still be retryable. + ok($node->psql('postgres', q(SELECT count(*) FROM repacked;)) == 0, + "$name: table readable after crash recovery"); + is( $node->safe_psql('postgres', + q[SELECT count(*) FROM pg_replication_slots;]), + 0, "$name: no leftover replication slot after restart"); + is( $node->safe_psql('postgres', + q[SELECT count(*) FROM pg_index WHERE NOT (indisvalid AND indisready);]), + 0, "$name: no leftover invalid index after restart"); + + $node->safe_psql('postgres', q(DROP INDEX IF EXISTS repacked_num2;)); + $node->safe_psql('postgres', q(REPACK (CONCURRENTLY) repacked;)); + + return; +}