From 119f39158c7b03f71e2fc6ecce9e9973eb995bbb Mon Sep 17 00:00:00 2001 From: Ayush Tiwari Date: Wed, 22 Jul 2026 05:15:44 +0000 Subject: [PATCH v5] Add pg_wal_preallocate() to eagerly create future WAL segments Write performance can depend heavily on whether a WAL segment is newly created or recycled. Creating a segment requires creating and zero-filling a temporary file (with wal_init_zero), fsyncing it, and then durably installing it. Recycling reuses an already allocated file and skips that initialization write and its explicit fsync. On non-copy-on-write file systems the difference can be significant, and it is paid by foreground backends on the write hot path. The server grows its pool of future WAL segments lazily, as WAL is produced. Right after initdb, or before a fresh benchmark or bulk load, that pool can be empty and backends create and fill segments themselves, causing latency spikes. This adds a superuser-only SQL function pg_wal_preallocate(bytes bigint DEFAULT NULL, force boolean DEFAULT false) returns bigint that samples the current WAL insertion location and pre-creates ceil(bytes / wal_segment_size) segment files, starting with the first unused segment at or after that location. It returns the number of files actually created. If bytes is omitted or NULL, min_wal_size is used as the requested byte count. An explicit request that would create more segments than fit within max_wal_size is rejected with an error rather than being silently reduced; setting force to true carries it out anyway. A request that omits bytes is not limited, since min_wal_size is what the server already tries to keep. The function uses the existing XLogFileInitInternal machinery. The slow initialization occurs without ControlFileLock, while final installation via InstallXLogFileSegment is synchronized with the checkpointer by that lock. The loop is interruptible, and the function refuses to run during recovery. This is deliberately a best-effort warm-up rather than an auto-tuning mechanism: concurrent WAL insertion can advance beyond the sampled location. Preallocated segments lie ahead of the insertion location, so a checkpoint does not remove them; the space is reclaimed only once WAL advances into them. On copy-on-write file systems, where recycling is not cheaper than creating (see wal_recycle), preallocation offers little benefit. --- doc/src/sgml/func/func-admin.sgml | 49 ++++++ src/backend/access/transam/xlog.c | 55 +++++++ src/backend/access/transam/xlogfuncs.c | 71 +++++++++ src/include/access/xlog.h | 1 + src/include/catalog/pg_proc.dat | 6 + src/test/recovery/meson.build | 1 + src/test/recovery/t/056_wal_preallocate.pl | 166 +++++++++++++++++++++ 7 files changed, 349 insertions(+) create mode 100644 src/test/recovery/t/056_wal_preallocate.pl diff --git a/doc/src/sgml/func/func-admin.sgml b/doc/src/sgml/func/func-admin.sgml index 0eae1c1f616..d9b5edbd87c 100644 --- a/doc/src/sgml/func/func-admin.sgml +++ b/doc/src/sgml/func/func-admin.sgml @@ -514,6 +514,55 @@ LOG: Grand total: 1651920 bytes in 201 blocks; 622360 free (88 chunks); 1029560 + + + + pg_wal_preallocate + + pg_wal_preallocate ( bytes bigint DEFAULT NULL, + force boolean DEFAULT false ) + bigint + + + Eagerly pre-creates + ceil(bytes / wal_segment_size) + write-ahead log segment files, starting with the first unused segment + at or after the current WAL insertion location, and returns the number + of files that were actually created. Segments that already exist are + not recreated, so this tops up the pool of ready future segments rather + than adding unconditionally, and the returned count may be smaller than + requested, possibly zero. If bytes is omitted or + NULL, the value of + is used, which keeps roughly that much ready ahead of the current + insertion location. This can be used to warm up the pool of future WAL + files before a burst of write activity, such as a benchmark run or a + bulk load, so that foreground WAL insertion does not have to create and + initialize new segment files itself. An explicit + bytes request that would create more segments + than fit within is rejected with an + error, unless force is true. + No such limit is applied when bytes is omitted or + NULL. + + + The current WAL insertion location is sampled when the function starts, + so concurrent WAL insertion can advance beyond that location while the + function runs. Preallocated files lie ahead of that location, so a + checkpoint does not remove them; they are reclaimed only once WAL + advances into them and a later checkpoint recycles or removes them. + Preallocating far beyond with + force can therefore keep a large amount of disk + space occupied for a long time. On file systems where recycling a WAL file + is not cheaper than creating a new one (for example copy-on-write file + systems, see ), preallocation provides + little benefit. This function cannot be executed during recovery. + + + This function is restricted to superusers by default, but other users + can be granted EXECUTE to run the function. + + + diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c index b23d8bbbdad..a4b3a5e02c8 100644 --- a/src/backend/access/transam/xlog.c +++ b/src/backend/access/transam/xlog.c @@ -3761,6 +3761,61 @@ PreallocXlogFiles(XLogRecPtr endptr, TimeLineID tli) } } +/* + * Eagerly pre-create up to 'nsegs' future WAL segments, starting with the + * first unused segment at or after the current WAL insertion location, and + * return the number of segments that were newly created. + * + * This backs the pg_wal_preallocate() SQL function. Unlike + * PreallocXlogFiles(), which lazily creates a single segment near the end of a + * checkpoint, this fills the future-segment pool on demand so that a + * subsequent burst of WAL activity does not pay the cost of creating and + * zero-filling segments in the foreground. It is a best-effort warm-up: + * segments created beyond what min_wal_size keeps may be recycled or removed + * again by a later checkpoint. + * + * The caller must not be in recovery. The work is interruptible, since + * creating many segments can take a while. + */ +int64 +PreallocXlogSegments(int64 nsegs) +{ + XLogSegNo segno; + XLogRecPtr insertptr; + TimeLineID tli; + int64 nsegsadded = 0; + + /* + * Bail out if we are in recovery, or if the startup process has disabled + * segment installation. The latter is an intentional unlocked read, as + * in PreallocXlogFiles(). + */ + if (RecoveryInProgress() || !XLogCtl->InstallXLogFileSegmentActive) + return 0; + + insertptr = GetXLogInsertRecPtr(); + tli = GetWALInsertionTimeLine(); + XLByteToPrevSeg(insertptr, segno, wal_segment_size); + + for (int64 i = 0; i < nsegs; i++) + { + bool added; + char path[MAXPGPATH]; + int lf; + + CHECK_FOR_INTERRUPTS(); + + segno++; + lf = XLogFileInitInternal(segno, tli, &added, path); + if (lf >= 0) + close(lf); + if (added) + nsegsadded++; + } + + return nsegsadded; +} + /* * Throws an error if the given log segment has already been removed or * recycled. The caller should only pass a segment that it knows to have diff --git a/src/backend/access/transam/xlogfuncs.c b/src/backend/access/transam/xlogfuncs.c index 1f52bf7b420..2e29a1e70ec 100644 --- a/src/backend/access/transam/xlogfuncs.c +++ b/src/backend/access/transam/xlogfuncs.c @@ -222,6 +222,77 @@ pg_switch_wal(PG_FUNCTION_ARGS) PG_RETURN_LSN(switchpoint); } +/* + * pg_wal_preallocate: eagerly pre-create future WAL segments + * + * Pre-creates ceil(bytes / wal_segment_size) WAL segments, starting with the + * first unused segment at or after the current WAL insertion location, and + * returns the number of segments that were newly created. If bytes is NULL + * (the default), min_wal_size is used. This lets an operator warm up the + * future-segment pool before a burst of write activity, so that foreground WAL + * insertion does not have to create and zero-fill segments itself. An + * explicit request larger than max_wal_size is rejected unless force is true. + * + * Permission checking for this function is managed through the normal GRANT + * system. + */ +Datum +pg_wal_preallocate(PG_FUNCTION_ARGS) +{ + int64 bytes; + bool force = !PG_ARGISNULL(1) && PG_GETARG_BOOL(1); + int64 nsegs; + int64 nsegsadded; + + if (RecoveryInProgress()) + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("recovery is in progress"), + errhint("WAL control functions cannot be executed during recovery."))); + + if (PG_ARGISNULL(0)) + bytes = (int64) min_wal_size_mb * 1024 * 1024; + else + { + bytes = PG_GETARG_INT64(0); + + if (bytes < 0) + ereport(ERROR, + (errcode(ERRCODE_NUMERIC_VALUE_OUT_OF_RANGE), + errmsg("number of bytes to preallocate must not be negative"))); + } + + /* Round up to a whole number of segments (overflow-safe). */ + nsegs = bytes / wal_segment_size; + if (bytes % wal_segment_size != 0) + nsegs++; + + /* + * Reject an explicit request that does not fit within max_wal_size, rather + * than silently carrying out something smaller than was asked for. The + * default request is not limited, since min_wal_size is what the server + * already tries to keep. + */ + if (!PG_ARGISNULL(0) && !force) + { + int64 maxsegs = XLogMBVarToSegs(max_wal_size_mb, + wal_segment_size); + + if (nsegs > maxsegs) + ereport(ERROR, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("WAL preallocation request exceeds \"max_wal_size\""), + errdetail("The request needs %lld WAL segments, but \"max_wal_size\" allows %lld.", + (long long) nsegs, (long long) maxsegs), + errhint("Call pg_wal_preallocate() with \"force\" set to true " + "to bypass this limit."))); + } + + nsegsadded = PreallocXlogSegments(nsegs); + + PG_RETURN_INT64(nsegsadded); +} + /* * pg_log_standby_snapshot: call LogStandbySnapshot() * diff --git a/src/include/access/xlog.h b/src/include/access/xlog.h index 4dd98624204..45db236ce6d 100644 --- a/src/include/access/xlog.h +++ b/src/include/access/xlog.h @@ -221,6 +221,7 @@ extern bool XLogBackgroundFlush(void); extern bool XLogNeedsFlush(XLogRecPtr record); extern int XLogFileInit(XLogSegNo logsegno, TimeLineID logtli); extern int XLogFileOpen(XLogSegNo segno, TimeLineID tli); +extern int64 PreallocXlogSegments(int64 nsegs); extern void CheckXLogRemoved(XLogSegNo segno, TimeLineID tli); extern XLogSegNo XLogGetLastRemovedSegno(void); diff --git a/src/include/catalog/pg_proc.dat b/src/include/catalog/pg_proc.dat index f8a021987b5..c90867eb63f 100644 --- a/src/include/catalog/pg_proc.dat +++ b/src/include/catalog/pg_proc.dat @@ -6809,6 +6809,12 @@ { oid => '2848', descr => 'switch to new wal file', proname => 'pg_switch_wal', provolatile => 'v', prorettype => 'pg_lsn', proargtypes => '', prosrc => 'pg_switch_wal', proacl => '{POSTGRES=X}' }, +{ oid => '9888', + descr => 'preallocate future WAL files, return number created', + proname => 'pg_wal_preallocate', provolatile => 'v', proisstrict => 'f', + prorettype => 'int8', proargtypes => 'int8 bool', + proargnames => '{bytes,force}', proargdefaults => '{NULL,false}', + prosrc => 'pg_wal_preallocate', proacl => '{POSTGRES=X}' }, { oid => '6305', descr => 'log details of the current snapshot to WAL', proname => 'pg_log_standby_snapshot', provolatile => 'v', prorettype => 'pg_lsn', proargtypes => '', diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build index 39ec8c4946d..61dcadd268d 100644 --- a/src/test/recovery/meson.build +++ b/src/test/recovery/meson.build @@ -64,6 +64,7 @@ tests += { 't/053_standby_login_event_trigger.pl', 't/054_unlogged_sequence_promotion.pl', 't/055_cascade_reconnect.pl', + 't/056_wal_preallocate.pl', ], }, } diff --git a/src/test/recovery/t/056_wal_preallocate.pl b/src/test/recovery/t/056_wal_preallocate.pl new file mode 100644 index 00000000000..61bf9af507b --- /dev/null +++ b/src/test/recovery/t/056_wal_preallocate.pl @@ -0,0 +1,166 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Test pg_wal_preallocate(), which eagerly creates future WAL segments to +# cover a requested number of bytes (defaulting to min_wal_size). +use strict; +use warnings FATAL => 'all'; + +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +# Count regular WAL segment files (24 hex digits) in a node's pg_wal. +sub count_segments +{ + my $node = shift; + my $waldir = $node->data_dir . '/pg_wal'; + opendir(my $dh, $waldir) or die "could not open $waldir: $!"; + my @segs = grep { /^[0-9A-F]{24}$/ } readdir($dh); + closedir($dh); + return scalar @segs; +} + +my $node = PostgreSQL::Test::Cluster->new('main'); +$node->init( + allows_streaming => 1, extra => ['--wal-segsize=16']); +# Keep the automatic recycled pool small and stable so the effect of +# preallocation is clearly attributable to the function. +$node->append_conf( + 'postgresql.conf', q{ +min_wal_size = 80MB +max_wal_size = 1GB +}); +$node->start; + +my $segsize = $node->safe_psql('postgres', + "SELECT pg_size_bytes(current_setting('wal_segment_size'))"); +note "wal_segment_size = $segsize bytes"; + +# Preallocate enough bytes for 10 segments; the pool should grow by the number +# reported, and (on a freshly started node) that number should be 10. +my $before = count_segments($node); +my $created = + $node->safe_psql('postgres', "SELECT pg_wal_preallocate(10 * $segsize)"); +is(count_segments($node), $before + $created, + 'pg_wal grew by the reported number of segments'); +is($created, 10, 'pg_wal_preallocate(10 segments worth) reports 10 created'); + +# A second identical call creates nothing new. +is($node->safe_psql('postgres', "SELECT pg_wal_preallocate(10 * $segsize)"), + 0, 'second call creates no additional segments'); + +# Zero bytes is a no-op. +is($node->safe_psql('postgres', 'SELECT pg_wal_preallocate(0)'), + 0, 'zero bytes creates nothing'); + +# A negative byte count is rejected. +my ($ret, $stdout, $stderr) = + $node->psql('postgres', 'SELECT pg_wal_preallocate(-1)'); +isnt($ret, 0, 'negative byte count errors out'); +like($stderr, qr/must not be negative/, + 'negative count produces the expected error'); + +# An explicit request larger than max_wal_size is rejected unless force is +# given. A request that omits bytes uses min_wal_size and is not limited, so +# min_wal_size is deliberately larger than max_wal_size here. +my $limit_node = PostgreSQL::Test::Cluster->new('limit'); +$limit_node->init(extra => ['--wal-segsize=16']); +$limit_node->append_conf( + 'postgresql.conf', q{ +min_wal_size = 80MB +max_wal_size = 65MB +}); +$limit_node->start; + +my $limit_segsize = $limit_node->safe_psql('postgres', + "SELECT pg_size_bytes(current_setting('wal_segment_size'))"); +is($limit_node->safe_psql('postgres', 'SELECT pg_wal_preallocate()'), + 5, 'default request is not limited by max_wal_size'); + +my $limit_before = count_segments($limit_node); +($ret, $stdout, $stderr) = $limit_node->psql( + 'postgres', "SELECT pg_wal_preallocate(6 * $limit_segsize)"); +isnt($ret, 0, 'explicit request above max_wal_size is rejected'); +like($stderr, qr/exceeds "max_wal_size"/, + 'rejected request reports the limit'); +is(count_segments($limit_node), $limit_before, + 'rejected request creates nothing'); + +is($limit_node->safe_psql( + 'postgres', "SELECT pg_wal_preallocate(6 * $limit_segsize, force => true)"), + 1, 'force carries out a request above max_wal_size'); +$limit_node->stop; + +# The no-argument form preallocates min_wal_size worth of segments, rounded up +# to a whole segment. Verify this with a non-aligned setting on a fresh node. +my $node2 = PostgreSQL::Test::Cluster->new('defaults'); +$node2->init(extra => ['--wal-segsize=16']); +$node2->append_conf( + 'postgresql.conf', q{ +min_wal_size = 33MB +max_wal_size = 1GB +}); +$node2->start; + +my $segsize2 = $node2->safe_psql('postgres', + "SELECT pg_size_bytes(current_setting('wal_segment_size'))"); +my $min_wal_bytes = $node2->safe_psql('postgres', + "SELECT pg_size_bytes(current_setting('min_wal_size'))"); +my $default_nsegs = int(($min_wal_bytes + $segsize2 - 1) / $segsize2); + +my $default_created = $node2->safe_psql('postgres', 'SELECT pg_wal_preallocate()'); +is($default_created, $default_nsegs, + 'no-argument call rounds min_wal_size up to whole segments'); + +# An explicit NULL is treated like the default (the function is not strict), so +# it also finds the pool already warm here. +is($node2->safe_psql('postgres', 'SELECT pg_wal_preallocate(NULL)'), + 0, 'explicit NULL uses the default'); +$node2->stop; + +# Byte counts are rounded up, and preallocation is anchored at the exact WAL +# insertion position. The restore point leaves that position within the first +# page after a segment switch; an approximate write pointer still identifies +# the previous segment there. +my $node3 = PostgreSQL::Test::Cluster->new('boundary'); +$node3->init( + allows_streaming => 1, extra => ['--wal-segsize=16']); +$node3->start; + +my $segsize3 = $node3->safe_psql('postgres', + "SELECT pg_size_bytes(current_setting('wal_segment_size'))"); +is($node3->safe_psql('postgres', 'SELECT pg_wal_preallocate(1)'), + 1, 'one byte rounds up to one segment'); + +$node3->safe_psql('postgres', q{ +SELECT pg_switch_wal(); +SELECT pg_create_restore_point('preallocation boundary test'); +}); +is($node3->safe_psql('postgres', "SELECT pg_wal_preallocate(4 * $segsize3)"), + 4, 'preallocation starts after the exact insertion segment'); +$node3->stop; + +# The function is superuser-only by default. +$node->safe_psql('postgres', 'CREATE ROLE regress_wp_user LOGIN'); +($ret, $stdout, $stderr) = $node->psql( + 'postgres', + 'SELECT pg_wal_preallocate(0)', + extra_params => [ '-U', 'regress_wp_user' ]); +isnt($ret, 0, 'non-superuser is denied by default'); +like($stderr, qr/permission denied/, 'permission denied for non-superuser'); + +# Recovery rejects the function even when the requested byte count is zero. +$node->backup('backup'); +my $standby = PostgreSQL::Test::Cluster->new('standby'); +$standby->init_from_backup($node, 'backup', has_streaming => 1); +$standby->start; +($ret, $stdout, $stderr) = + $standby->psql('postgres', 'SELECT pg_wal_preallocate(0)'); +isnt($ret, 0, 'preallocation is rejected during recovery'); +like($stderr, qr/recovery is in progress/, + 'recovery rejection produces the expected error'); +$standby->stop; + +$node->stop; + +done_testing(); -- 2.43.0