From 924d5bbb4698ceba38f6468f1bcdc0c069a6a6c0 Mon Sep 17 00:00:00 2001 From: Melanie Plageman Date: Wed, 23 Sep 2026 11:59:53 -0400 Subject: [PATCH v3 1/3] Read visibility map pages with RBM_ZERO_ON_ERROR in VM clear redo ed62d26caca started registering VM blocks when clearing the VM which is required for protection against torn pages as well as for correct incremental backups. However, it read the VM pages in recovery with RBM_NORMAL which errors out when it encounters a corrupt page. This is usually desirable, however, we still retain code paths that modify the VM in recovery without the block having been registered. A crash while modifying the VM page could lead to a corrupt page and no FPI to recover it. As long as we can trivially produce corrupt pages during recovery through our own redo mechanism, we shouldn't error out when reading a corrupt VM page. Make clearing the VM read the page with RBM_ZERO_ON_ERROR. This is consistent with the VM's other redo paths which set the VM bit (heap_xlog_prune_freeze() and heap_xlog_multi_insert()). Backpatch-through: 17 --- Notes: This is v2-0002 from https://postgr.es/m/CAAKRu_bApoksLDb-HX0GYciU3uLWqA1JagntaV8GP0%3D%2BidehHw%40mail.gmail.com The commit message is the same. The code is adjusted to apply to REL_19_STABLE without v2-0001. It fixes the standby PANIC that Jacky Nguyen reported in https://postgr.es/m/CAL4mQLAp562c1rCgg2Dqx6TBSdOk6vOvaLFjwFG9E6k4uwvJJw@mail.gmail.com src/backend/access/heap/heapam_xlog.c | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/src/backend/access/heap/heapam_xlog.c b/src/backend/access/heap/heapam_xlog.c index fae3b477c09..288b4952579 100644 --- a/src/backend/access/heap/heapam_xlog.c +++ b/src/backend/access/heap/heapam_xlog.c @@ -53,8 +53,9 @@ heap_xlog_vm_clear(XLogReaderState *record, */ if (XLogRecHasBlockRef(record, wal_vm_block_id)) { - if (XLogReadBufferForRedo(record, wal_vm_block_id, - &vmbuffer) == BLK_NEEDS_REDO) + if (XLogReadBufferForRedoExtended(record, wal_vm_block_id, + RBM_ZERO_ON_ERROR, false, + &vmbuffer) == BLK_NEEDS_REDO) { if (visibilitymap_clear(reln, heap_blkno, vmbuffer, flags)) PageSetLSN(BufferGetPage(vmbuffer), lsn); @@ -819,8 +820,9 @@ heap_xlog_update(XLogReaderState *record, bool hot_update) Assert(xlrec->flags & XLH_UPDATE_NEW_ALL_VISIBLE_CLEARED); - if (XLogReadBufferForRedo(record, HEAP_UPDATE_BLKREF_VM_NEW, - &vmbuffer_new) == BLK_NEEDS_REDO) + if (XLogReadBufferForRedoExtended(record, HEAP_UPDATE_BLKREF_VM_NEW, + RBM_ZERO_ON_ERROR, false, + &vmbuffer_new) == BLK_NEEDS_REDO) { /* * If both the old and new heap pages were all-visible and their @@ -857,8 +859,9 @@ heap_xlog_update(XLogReaderState *record, bool hot_update) Assert(xlrec->flags & XLH_UPDATE_OLD_ALL_VISIBLE_CLEARED); - if (XLogReadBufferForRedo(record, HEAP_UPDATE_BLKREF_VM_OLD, - &vmbuffer_old) == BLK_NEEDS_REDO) + if (XLogReadBufferForRedoExtended(record, HEAP_UPDATE_BLKREF_VM_OLD, + RBM_ZERO_ON_ERROR, false, + &vmbuffer_old) == BLK_NEEDS_REDO) { if (visibilitymap_clear(reln, oldblk, vmbuffer_old, VISIBILITYMAP_VALID_BITS)) -- 2.37.1 (Apple Git-137.1) From 0b36adcefe72b5c7197f95559df5364e51827388 Mon Sep 17 00:00:00 2001 From: Rahul Yadav Date: Thu, 1 Oct 2026 15:07:40 +0000 Subject: [PATCH v3 2/3] Initialize zeroed VM pages in VM clear redo RBM_ZERO_ON_ERROR recreates a truncated VM page as all zeros. With wal_consistency_checking, verifyBackupPageConsistency() then masks it with heap_mask(), and mask_unused_space() rejects a page with pd_lower 0. heap_xlog_prune_freeze() and heap_xlog_multi_insert() already initialize a VM page that was read as zeros. Do the same at the three VM clear sites. Discussion: https://postgr.es/m/P2s-NV0--F-9@rhyadav.dev --- Notes: Rahul posted this as a diff on top of v2, without a commit message. The message above is put together from his mail. src/backend/access/heap/heapam_xlog.c | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/src/backend/access/heap/heapam_xlog.c b/src/backend/access/heap/heapam_xlog.c index 288b4952579..1ce0c09a083 100644 --- a/src/backend/access/heap/heapam_xlog.c +++ b/src/backend/access/heap/heapam_xlog.c @@ -57,6 +57,10 @@ heap_xlog_vm_clear(XLogReaderState *record, RBM_ZERO_ON_ERROR, false, &vmbuffer) == BLK_NEEDS_REDO) { + /* initialize the page if it was read as zeros */ + if (PageIsNew(BufferGetPage(vmbuffer))) + PageInit(BufferGetPage(vmbuffer), BLCKSZ, 0); + if (visibilitymap_clear(reln, heap_blkno, vmbuffer, flags)) PageSetLSN(BufferGetPage(vmbuffer), lsn); } @@ -824,6 +828,10 @@ heap_xlog_update(XLogReaderState *record, bool hot_update) RBM_ZERO_ON_ERROR, false, &vmbuffer_new) == BLK_NEEDS_REDO) { + /* initialize the page if it was read as zeros */ + if (PageIsNew(BufferGetPage(vmbuffer_new))) + PageInit(BufferGetPage(vmbuffer_new), BLCKSZ, 0); + /* * If both the old and new heap pages were all-visible and their * VM bits are on the same VM page, that single VM page is @@ -863,6 +871,10 @@ heap_xlog_update(XLogReaderState *record, bool hot_update) RBM_ZERO_ON_ERROR, false, &vmbuffer_old) == BLK_NEEDS_REDO) { + /* initialize the page if it was read as zeros */ + if (PageIsNew(BufferGetPage(vmbuffer_old))) + PageInit(BufferGetPage(vmbuffer_old), BLCKSZ, 0); + if (visibilitymap_clear(reln, oldblk, vmbuffer_old, VISIBILITYMAP_VALID_BITS)) PageSetLSN(BufferGetPage(vmbuffer_old), lsn); -- 2.37.1 (Apple Git-137.1) From 364c4c54a128f698fff8c2b9b7ae26b55378dcf4 Mon Sep 17 00:00:00 2001 From: Shihao Date: Thu, 1 Oct 2026 22:38:25 -0600 Subject: [PATCH v3 3/3] Add test for standby restart after VM truncation With full_page_writes off, redo of a record that clears VM bits does not restore the VM page from an image. The test clears VM bits by delete, same-page update and cross-page update, truncates the tables, and restarts the standby from a restartpoint taken before those changes. wal_consistency_checking is on, so the VM page that redo recreates is checked too. Author: Shihao Zhong Discussion: https://postgr.es/m/CAL4mQLAp562c1rCgg2Dqx6TBSdOk6vOvaLFjwFG9E6k4uwvJJw@mail.gmail.com --- src/test/recovery/meson.build | 1 + src/test/recovery/t/058_vm_clear_truncate.pl | 88 ++++++++++++++++++++ 2 files changed, 89 insertions(+) create mode 100644 src/test/recovery/t/058_vm_clear_truncate.pl diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build index ebb12dd8766..bb28f9cfb81 100644 --- a/src/test/recovery/meson.build +++ b/src/test/recovery/meson.build @@ -66,6 +66,7 @@ tests += { 't/055_cascade_reconnect.pl', 't/056_standby_snapshot_export.pl', 't/057_snapshot_commit_race.pl', + 't/058_vm_clear_truncate.pl', ], }, } diff --git a/src/test/recovery/t/058_vm_clear_truncate.pl b/src/test/recovery/t/058_vm_clear_truncate.pl new file mode 100644 index 00000000000..294eb7de95e --- /dev/null +++ b/src/test/recovery/t/058_vm_clear_truncate.pl @@ -0,0 +1,88 @@ + +# Copyright (c) 2026, PostgreSQL Global Development Group + +# A standby must be able to restart when WAL it replays again clears +# visibility map bits on a VM page that a later, already replayed, +# truncation removed. With full_page_writes off, redo cannot restore the +# VM page from an image in the clearing record, so it has to cope with the +# page not existing. wal_consistency_checking is on so that the page redo +# recreates is checked as well. +use strict; +use warnings FATAL => 'all'; +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +my $primary = PostgreSQL::Test::Cluster->new('primary'); +$primary->init(allows_streaming => 1); +$primary->append_conf( + 'postgresql.conf', qq{ +full_page_writes = off +wal_consistency_checking = all +autovacuum = off +}); +$primary->start; +$primary->backup('bkp'); + +my $standby = PostgreSQL::Test::Cluster->new('standby'); +$standby->init_from_backup($primary, 'bkp', has_streaming => 1); +$standby->start; + +# Make every heap page all-visible, then make the standby create a +# restartpoint, so that a restart replays the changes below again. The +# row lock clears the all-frozen bit of vm_upd's first page. Otherwise +# the cross-page update below would first log a lock record that clears +# that bit, and the update would not be the first record to read the VM +# page. +$primary->safe_psql( + 'postgres', q{ +CREATE TABLE vm_del (a int); +CREATE TABLE vm_hot (a int) WITH (fillfactor = 50); +CREATE TABLE vm_upd (a int); +INSERT INTO vm_del SELECT generate_series(1, 1000); +INSERT INTO vm_hot SELECT generate_series(1, 1000); +INSERT INTO vm_upd SELECT generate_series(1, 1000); +VACUUM (FREEZE) vm_del, vm_hot, vm_upd; +SELECT a FROM vm_upd WHERE a = 1 FOR UPDATE; +CHECKPOINT; +}); +$primary->wait_for_replay_catchup($standby); +$standby->safe_psql('postgres', 'CHECKPOINT'); + +my $start_lsn = + $primary->safe_psql('postgres', 'SELECT pg_current_wal_insert_lsn()'); + +# Clear VM bits through delete, same-page update (old VM block) and +# cross-page update (new VM block), then truncate all three tables to +# zero blocks. +$primary->safe_psql( + 'postgres', q{ +DELETE FROM vm_del; +UPDATE vm_hot SET a = -a WHERE a = 1; +UPDATE vm_upd SET a = -a WHERE a = 1; +DELETE FROM vm_hot; +DELETE FROM vm_upd; +VACUUM vm_del, vm_hot, vm_upd; +}); +is( $primary->safe_psql( + 'postgres', + "SELECT sum(pg_relation_size(c, 'vm')) FROM unnest('{vm_del,vm_hot,vm_upd}'::regclass[]) c" + ), + '0', + 'VMs truncated on primary'); +$primary->wait_for_replay_catchup($standby); + +$standby->stop; +my $log_offset = -s $standby->logfile; +my $ret = $standby->start(fail_ok => 1); + +my $log = slurp_file($standby->logfile, $log_offset); +my ($redo_lsn) = $log =~ /redo starts at ([0-9A-F]+\/[0-9A-F]+)/; +ok( defined($redo_lsn) + && $primary->safe_psql('postgres', + "SELECT '$redo_lsn'::pg_lsn < '$start_lsn'::pg_lsn") eq 't', + 'redo after restart starts before the VM bits were cleared'); +ok($ret, 'standby restarts after replaying VM truncation'); +unlike($log, qr/invalid pages/, 'no invalid page references in standby log'); + +done_testing(); -- 2.37.1 (Apple Git-137.1)