From 1e5168bc01a65b572cec1035284326cec0ede337 Mon Sep 17 00:00:00 2001 From: Srinath Reddy Sadipiralla Date: Sat, 10 Oct 2026 12:09:27 +0530 Subject: [PATCH v1 2/2] Add a TAP test and documentation for fast crash recovery The test crashes a server with changes that exist only in WAL, including a table created and dropped inside the window, recovers with fast_crash_recovery on, and checks the data by sequential and index scan right after the server opens, after the worker has drained the index, and after a clean restart that must need no recovery. Document the setting, the new process type. --- doc/src/sgml/config.sgml | 30 +++++ doc/src/sgml/monitoring.sgml | 7 + src/test/recovery/meson.build | 1 + .../recovery/t/060_fast_crash_recovery.pl | 122 ++++++++++++++++++ 4 files changed, 160 insertions(+) create mode 100644 src/test/recovery/t/060_fast_crash_recovery.pl diff --git a/doc/src/sgml/config.sgml b/doc/src/sgml/config.sgml index e537b3b0043..218469f9da2 100644 --- a/doc/src/sgml/config.sgml +++ b/doc/src/sgml/config.sgml @@ -4246,6 +4246,35 @@ include_dir 'conf.d' + + fast_crash_recovery (boolean) + + fast_crash_recovery configuration parameter + + + + + Replay WAL on demand after a crash, instead of before accepting + connections. When enabled, crash recovery scans the WAL written since + the last checkpoint, replays only the records that do not modify + relation pages, and remembers which records apply to each page. The + server then accepts connections, and a page is recovered the first + time it is read from disk, before any process can see it. A + background worker recovers the remaining pages; until it has finished, + checkpoints are postponed and a clean shutdown waits for it. + The default is off. + This parameter can only be set at server start. + + + This shortens the time until the server accepts connections after a + crash, in exchange for slower first reads of pages that still need + recovery, and for dynamic shared memory proportional to the amount of + WAL to recover. It applies to crash recovery only: archive recovery, + standby mode and single-user mode replay WAL as usual. + + + + @@ -7513,6 +7542,7 @@ local0.* /var/log/postgresql bgwriter checkpointer checksums + fastrecoveryworker ioworker postmaster slotsyncworker diff --git a/doc/src/sgml/monitoring.sgml b/doc/src/sgml/monitoring.sgml index 6337d2a3d25..735b1bc2a88 100644 --- a/doc/src/sgml/monitoring.sgml +++ b/doc/src/sgml/monitoring.sgml @@ -1110,6 +1110,13 @@ postgres 27093 0.0 0.0 30096 2752 ? Ss 11:34 0:00 postgres: ser calculates data checksums for all pages in one database. + + + fast recovery worker: The background process that + recovers the pages still pending after a crash recovery with + enabled. + + io worker: A background process performing diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build index e1cf57ff611..80938652c32 100644 --- a/src/test/recovery/meson.build +++ b/src/test/recovery/meson.build @@ -68,6 +68,7 @@ tests += { 't/057_snapshot_commit_race.pl', 't/058_shutdown_crash_restart.pl', 't/059_remote_apply_status_interval.pl', + 't/060_fast_crash_recovery.pl', ], }, } diff --git a/src/test/recovery/t/060_fast_crash_recovery.pl b/src/test/recovery/t/060_fast_crash_recovery.pl new file mode 100644 index 00000000000..3171ca80a1e --- /dev/null +++ b/src/test/recovery/t/060_fast_crash_recovery.pl @@ -0,0 +1,122 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Crash recovery with on-demand WAL replay (fast_crash_recovery): the server +# accepts connections once the WAL has been scanned, each page is recovered +# when it is first read, and a background worker recovers the rest. The +# data must be right at every stage: right after the server opens, after the +# worker has drained the index, and after a clean restart. + +use strict; +use warnings FATAL => 'all'; +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +my $node = PostgreSQL::Test::Cluster->new('primary'); +$node->init; +$node->append_conf( + 'postgresql.conf', qq( +fast_crash_recovery = on +checkpoint_timeout = 1d +max_wal_size = 10GB +log_checkpoints = on +)); +$node->start; + +# A table with an index, checkpointed, then modified so that the changes +# exist only in WAL when we crash: updates, deletes, inserts, and a table +# that is created and dropped inside the window. +# +# A second table is truncated by VACUUM inside the window. The truncation +# record has no block reference, so the startup process replays it during +# the scan, and doing so reads the last visibility map and free space map +# page into shared buffers to clear their tails. The later changes to +# all-visible pages reference that visibility map page, so they test what +# happens to records for a page that is resident during the scan; and the +# table grows past the truncation point again, so a truncation replayed out +# of order would lose those rows. +$node->safe_psql( + 'postgres', q( +CREATE TABLE t (id int PRIMARY KEY, v int NOT NULL, pad text); +INSERT INTO t SELECT g, 0, repeat('x', 200) FROM generate_series(1, 20000) g; +CREATE TABLE shrunk (id int PRIMARY KEY, v int NOT NULL, pad text); +INSERT INTO shrunk SELECT g, 0, repeat('y', 500) FROM generate_series(1, 4000) g; +VACUUM shrunk; +CHECKPOINT; +UPDATE t SET v = v + 1 WHERE id % 3 = 0; +DELETE FROM t WHERE id % 10 = 0; +INSERT INTO t SELECT g, -1, '' FROM generate_series(20001, 25000) g; +CREATE TABLE dropped AS SELECT g FROM generate_series(1, 5000) g; +DROP TABLE dropped; +VACUUM t; +DELETE FROM shrunk WHERE id > 2000; +VACUUM shrunk; +UPDATE shrunk SET v = 1 WHERE id IN (1, 500, 1000, 1999); +INSERT INTO shrunk SELECT g, 2, '' FROM generate_series(2001, 2500) g; +)); + +my $seqscan = 'SELECT count(*), sum(v), sum(length(pad)) FROM t'; +my $indexscan = + 'SET enable_seqscan = off; SET enable_bitmapscan = off; ' + . 'SELECT count(*), sum(v) FROM t WHERE id > 0'; +my $shrunkscan = + "SELECT count(*), sum(v), pg_relation_size('shrunk') FROM shrunk"; +my $expected_seq = $node->safe_psql('postgres', $seqscan); +my $expected_idx = $node->safe_psql('postgres', $indexscan); +my $expected_shrunk = $node->safe_psql('postgres', $shrunkscan); + +# Crash. Log at DEBUG1 from here on, to see the scan evict the pages the +# truncation read. +$node->append_conf('postgresql.conf', 'log_min_messages = debug1'); +$node->stop('immediate'); +my $log_offset = -s $node->logfile; +$node->start; + +ok( $node->log_contains('fast recovery worker started', $log_offset), + 'crash recovery used on-demand WAL replay'); +ok( $node->log_contains( + qr/fast recovery: evicted block \d+ of relation \S+ fork \d+ to index its records/, + $log_offset), + 'a page read during the scan was evicted to index its later records'); + +# 1. Right after the server opened: pages are replayed as they are read, by +# a sequential scan and by an index scan. +is($node->safe_psql('postgres', $seqscan), + $expected_seq, 'data correct right after the server opened (seq scan)'); +is($node->safe_psql('postgres', $indexscan), + $expected_idx, 'data correct right after the server opened (index scan)'); +is($node->safe_psql('postgres', $shrunkscan), + $expected_shrunk, + 'truncated-and-regrown table correct right after the server opened'); + +# 2. After the worker has recovered every remaining page. +$node->wait_for_log(qr/fast crash recovery complete/, $log_offset); +is($node->safe_psql('postgres', $seqscan), + $expected_seq, 'data correct after the worker drained the index'); +is($node->safe_psql('postgres', $shrunkscan), + $expected_shrunk, + 'truncated-and-regrown table correct after the worker drained the index'); + +# Checkpoints are allowed again once the index is gone. +$node->safe_psql('postgres', 'CHECKPOINT'); +ok( $node->log_contains('checkpoint complete', $log_offset), + 'a checkpoint ran after the index was drained'); + +# 3. After a clean restart, which must not need recovery at all. +$log_offset = -s $node->logfile; +$node->restart; +ok( !$node->log_contains('automatic recovery in progress', $log_offset), + 'clean restart needed no recovery'); +is($node->safe_psql('postgres', $seqscan), + $expected_seq, 'data correct after a clean restart'); +is($node->safe_psql('postgres', $shrunkscan), + $expected_shrunk, 'truncated-and-regrown table correct after a clean restart'); + +# The dropped table's pages were forgotten, not resurrected. +is( $node->safe_psql( + 'postgres', "SELECT count(*) FROM pg_class WHERE relname = 'dropped'"), + '0', + 'table dropped inside the window stays dropped'); + +$node->stop; +done_testing(); -- 2.43.0