From d2f332d07678f77a8dd9bb24b4a14eedc9ff3290 Mon Sep 17 00:00:00 2001 From: Srinath Reddy Sadipiralla Date: Sat, 10 Oct 2026 12:09:27 +0530 Subject: [PATCH v1 1/2] Replay WAL on demand after a crash instead of before accepting connections After a crash, the server cannot accept connections until every WAL record written since the last checkpoint has been replayed and, in crash recovery, until the end-of-recovery checkpoint has written the result back. Both are proportional to the amount of WAL, so availability after a crash is bound by WAL volume. With fast_crash_recovery enabled, the startup process scans that WAL once but applies only the records that do not modify relation pages: transaction status, multixacts, relation creation, truncation and removal, database operations, and so on, which are cheap and have no page read that could trigger them later. Records that do modify pages are not applied; the startup process remembers, per page, the LSNs of the records that touch it, in a dshash table in dynamic shared memory. The server then accepts connections. A page is recovered the first time any process reads it from disk: the buffer manager reads it under its I/O-in-progress flag, replays its pending records, and only then marks it valid, so no process can see the page before it is current, and the others wait as they would behind a slow disk. A new auxiliary process, the fast recovery worker, reads every page that is still pending, so the system converges; once it is done it destroys the index and requests a checkpoint. Replaying a page's records applies only the parts of each record that belong to that page: the record's other blocks report BLK_DONE, which redo routines must cope with anyway because any block can already be past a record's LSN after a crash during recovery; blocks a record initializes get a scratch local buffer, and blocks with a full-page image get the image restored into one, since redo routines expect BLK_RESTORED for those. Redo is handed the buffer being replayed into, with an exclusive lock standing in for a cleanup lock on that buffer only, since nobody can hold pointers into a page that has never been valid. A page is replayed exactly once: its entries are retired when it becomes valid, and pages about to be zeroed forget theirs. Redo of an indexed page reads no other page, so replays never nest. InRecovery is set for the duration of on-demand redo, which redo routines rely on. If a replay fails, the transaction abort path (or process exit, for a FATAL) cleans the buffer, which redo dirtied before it was valid, and resets the replay state, the way pgaio_error_cleanup() undoes AIO state; the resource owner then ends the I/O as for any failed read, and the next reader replays the page from scratch. Checkpoints are a promise that everything before the redo point is on disk, and pending pages would break it, so CreateCheckPoint() refuses while the index exists, unwaited requests are postponed, and a clean shutdown waits for the worker. A truncation is the one eager record that reads relation pages (the last FSM and VM page, to clear their tails); before a later record for such a page is indexed, the startup process writes the page out and drops it from shared buffers, since a resident page is never read from disk again and its pending records would never be applied. Relations and databases dropped or truncated later in the WAL forget their pending pages. The startup process extends a relation's file to cover any block it indexes beyond the file's end, as stock recovery would while replaying, so relation sizes are true before the server opens. Fast crash recovery applies to crash recovery under the postmaster only; archive recovery, standby mode and single-user mode replay WAL as usual. Known limitations, to be addressed separately: the index is not yet bounded in size; pages with pending records are read synchronously, one at a time, rather than through combined asynchronous I/O; the worker replays pages serially. --- src/backend/access/transam/Makefile | 3 +- src/backend/access/transam/lsn_indexer.c | 1094 +++++++++++++++++ src/backend/access/transam/meson.build | 1 + src/backend/access/transam/xact.c | 3 + src/backend/access/transam/xlog.c | 39 +- src/backend/access/transam/xlogrecovery.c | 85 +- src/backend/access/transam/xlogutils.c | 88 +- src/backend/catalog/storage.c | 6 + src/backend/postmaster/Makefile | 3 +- src/backend/postmaster/checkpointer.c | 61 + src/backend/postmaster/fast_recovery_worker.c | 143 +++ src/backend/postmaster/launch_backend.c | 1 + src/backend/postmaster/meson.build | 1 + src/backend/postmaster/pmchild.c | 1 + src/backend/postmaster/postmaster.c | 34 + src/backend/storage/buffer/bufmgr.c | 115 ++ src/backend/storage/buffer/localbuf.c | 8 +- src/backend/storage/freespace/freespace.c | 12 + src/backend/storage/ipc/ipci.c | 1 + src/backend/utils/activity/pgstat_backend.c | 1 + src/backend/utils/activity/pgstat_io.c | 24 +- .../utils/activity/wait_event_names.txt | 4 + src/backend/utils/misc/guc_parameters.dat | 7 + src/backend/utils/misc/guc_tables.c | 1 + src/backend/utils/misc/postgresql.conf.sample | 5 + src/include/access/lsn_indexer.h | 142 +++ src/include/access/xlog.h | 1 + src/include/miscadmin.h | 1 + src/include/postmaster/fast_recovery_worker.h | 17 + src/include/postmaster/postmaster.h | 2 + src/include/postmaster/proctypelist.h | 1 + src/include/storage/bufmgr.h | 3 + src/include/storage/lwlocklist.h | 3 + src/include/storage/subsystemlist.h | 1 + src/test/regress/expected/stats.out | 5 +- src/tools/pgindent/typedefs.list | 3 + 36 files changed, 1901 insertions(+), 19 deletions(-) create mode 100644 src/backend/access/transam/lsn_indexer.c create mode 100644 src/backend/postmaster/fast_recovery_worker.c create mode 100644 src/include/access/lsn_indexer.h create mode 100644 src/include/postmaster/fast_recovery_worker.h diff --git a/src/backend/access/transam/Makefile b/src/backend/access/transam/Makefile index a32f473e0a2..87ce11cc970 100644 --- a/src/backend/access/transam/Makefile +++ b/src/backend/access/transam/Makefile @@ -37,7 +37,8 @@ OBJS = \ xlogrecovery.o \ xlogstats.o \ xlogutils.o \ - xlogwait.o + xlogwait.o \ + lsn_indexer.o include $(top_srcdir)/src/backend/common.mk diff --git a/src/backend/access/transam/lsn_indexer.c b/src/backend/access/transam/lsn_indexer.c new file mode 100644 index 00000000000..dcb06d83fd2 --- /dev/null +++ b/src/backend/access/transam/lsn_indexer.c @@ -0,0 +1,1094 @@ +/*------------------------------------------------------------------------- + * + * lsn_indexer.c + * Index of pending WAL records by page, for on-demand replay after a crash + * + * During crash recovery the startup process scans the WAL and, instead of + * replaying the records that modify relation pages, files the LSN of each + * one under the page it touches, in a dshash table in dynamic shared memory. + * The server then opens. A page's pending records are replayed when the + * page is first read from disk, by whichever process reads it, and the fast + * recovery worker reads the rest in the background. + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * + * IDENTIFICATION + * src/backend/access/transam/lsn_indexer.c + * + *------------------------------------------------------------------------- + */ +#include "postgres.h" + +#include "access/lsn_indexer.h" +#include "access/xlog.h" +#include "access/xlog_internal.h" +#include "access/xlogreader.h" +#include "access/xlogrecovery.h" +#include "access/xlogutils.h" +#include "common/hashfn.h" +#include "common/relpath.h" +#include "lib/dshash.h" +#include "lib/stringinfo.h" +#include "miscadmin.h" +#include "storage/buf_internals.h" +#include "storage/bufmgr.h" +#include "storage/ipc.h" +#include "storage/latch.h" +#include "storage/lwlock.h" +#include "storage/shmem.h" +#include "storage/smgr.h" +#include "utils/dsa.h" +#include "utils/hsearch.h" +#include "utils/memutils.h" +#include "utils/wait_event.h" + +/* GUC variable */ +bool fast_crash_recovery = false; + +/* Control structure in traditional shared memory */ +LSNIndexControl *LSNIndexCtl = NULL; + +/* Per-process DSA and dshash handles (set by LSNIndexAttach) */ +static dsa_area *lsn_dsa = NULL; +static dshash_table *lsn_hash = NULL; + +/* State for on-demand replay; see lsn_indexer.h */ +BufferTag targetTag; +bool inReplayPageWals = false; +Buffer targetBuffer = InvalidBuffer; + +/* + * The page this process is replaying on demand, between + * LSNIndexBeginPageReplay() and LSNIndexEndPageReplay(). Replays never + * nest: redo of an indexed page touches no other page (see + * LSNIndexBeginPageReplay()), so one slot is enough. + */ +static bool replay_begun = false; +static BufferTag replaying_tag; + +/* InRecovery before on-demand redo set it; see LSNIndexReplayIntoBuffer() */ +static bool saved_in_recovery = false; + +/* Is lsn_index_exit_callback() registered in this process? */ +static bool exit_callback_registered = false; + +/* Have we run RmgrStartup() for on-demand replay in this process? */ +static bool rmgrs_started = false; + +/* + * Scratch buffers handed out for the current record's other pages; dropped + * once the record has been replayed. A record references at most + * XLR_MAX_BLOCK_ID + 1 blocks. + */ +static BufferTag scratch_tags[XLR_MAX_BLOCK_ID + 1]; +static int nscratch = 0; + +/* What the scan did; reported by LSNIndexLogScanSummary() */ +static int64 scan_records_indexed = 0; +static int64 scan_pages_indexed = 0; +static int64 scan_records_replayed = 0; /* with block references */ +static int64 scan_pages_evicted = 0; +static int64 scan_blocks_extended = 0; + +/* + * dshash parameters for the LSN index. + * + * Key = BufferTag (first field of PageLSNEntry). + * Built-in memcmp/memhash/memcpy work fine for BufferTag. + */ +static const dshash_parameters lsn_dsh_params = { + sizeof(BufferTag), /* key_size */ + sizeof(PageLSNEntry), /* entry_size */ + dshash_memcmp, /* compare_function */ + dshash_memhash, /* hash_function */ + dshash_memcpy, /* copy_function */ + LWTRANCHE_LSN_INDEX_HASH /* tranche_id */ +}; + +static void LSNIndexShmemRequest(void *arg); + +const ShmemCallbacks LsnIndexerShmemCallbacks = { + .request_fn = LSNIndexShmemRequest, + .init_fn = LSNIndexShmemInit, +}; + +/* ---------------------------------------------------------------- + * Shared-memory sizing and initialisation + * ---------------------------------------------------------------- + */ + +static void +LSNIndexShmemRequest(void *arg) +{ + Size size; + + size = MAXALIGN(sizeof(LSNIndexControl)); + + if (!fast_crash_recovery) + return; + ShmemRequestStruct(.name = "LSNIndex Ctl", + .size = size, + .ptr = (void **) &LSNIndexCtl, + ); +} + +/* + * LSNIndexShmemInit - postmaster-safe phase 1 of init. + * + * Allocates the small control struct in main shmem and initializes the + * DSA/dshash handles to "invalid". This must be safe to run in the + * postmaster, so we deliberately do NOT call dsa_create here - dsm.c + * forbids that with an assertion (no PROC entry, no error path, etc.). + * + * Children inherit the LSNIndexCtl pointer through fork. Whichever + * process needs the index first calls LSNIndexInit() to materialize the + * DSA+dshash and publish their handles via LSNIndexCtl. + * + * Called from CreateOrAttachShmemStructs. + */ +void +LSNIndexShmemInit(void *arg) +{ + if (!fast_crash_recovery) + return; + + LSNIndexCtl->dsa_handle = DSA_HANDLE_INVALID; + LSNIndexCtl->hash_handle = DSHASH_HANDLE_INVALID; + LSNIndexCtl->is_active = false; + pg_atomic_init_u32(&LSNIndexCtl->nreplaying, 0); +} + +/* ---------------------------------------------------------------- + * Lazy create + attach + * ---------------------------------------------------------------- + */ + +/* + * Run LSNIndexErrorCleanup() if this process exits with a replay in + * progress. An ERROR in an auxiliary process is a FATAL, and a FATAL never + * reaches the transaction abort path, where the cleanup normally runs. + * before_shmem_exit() callbacks run last-registered first, and we register + * after ShutdownPostgres() and ReleaseAuxProcessResources(), the two that + * would otherwise reach AbortBufferIO() with the buffer still dirty. + */ +static void +lsn_index_exit_callback(int code, Datum arg) +{ + LSNIndexErrorCleanup(); +} + +static void +register_exit_callback(void) +{ + if (exit_callback_registered) + return; + before_shmem_exit(lsn_index_exit_callback, 0); + exit_callback_registered = true; +} + +/* + * LSNIndexInit - phase 2: create the DSA+dshash if not yet done, else attach. + * + * Must run in a backend context (the startup process is fine). Called by + * the startup process when it begins crash recovery. Pins the DSA so it + * survives past startup-process exit; subsequent attachers (the worker and + * regular backends) just call LSNIndexAttach(). + * + * Modeled on init_dsm_registry() in src/backend/storage/ipc/dsm_registry.c. + */ +void +LSNIndexInit(void) +{ + MemoryContext oldcxt; + + /* Quick exit if already attached in this process. */ + if (lsn_hash != NULL) + return; + + Assert(LSNIndexCtl != NULL); + + oldcxt = MemoryContextSwitchTo(TopMemoryContext); + + LWLockAcquire(LSNIndexLock, LW_EXCLUSIVE); + + if (LSNIndexCtl->dsa_handle == DSA_HANDLE_INVALID) + { + /* First in - create the DSA + dshash and publish handles. */ + lsn_dsa = dsa_create(LWTRANCHE_LSN_INDEX_DSA); + dsa_pin(lsn_dsa); + dsa_pin_mapping(lsn_dsa); + + lsn_hash = dshash_create(lsn_dsa, &lsn_dsh_params, NULL); + + LSNIndexCtl->dsa_handle = dsa_get_handle(lsn_dsa); + LSNIndexCtl->hash_handle = dshash_get_hash_table_handle(lsn_hash); + LSNIndexCtl->is_active = true; + } + else + { + /* Already created - just attach. */ + lsn_dsa = dsa_attach(LSNIndexCtl->dsa_handle); + dsa_pin_mapping(lsn_dsa); + lsn_hash = dshash_attach(lsn_dsa, &lsn_dsh_params, + LSNIndexCtl->hash_handle, NULL); + } + + LWLockRelease(LSNIndexLock); + MemoryContextSwitchTo(oldcxt); + + register_exit_callback(); +} + +/* + * LSNIndexAttach - attach to an already-created index. + * + * Used by backends and the fast recovery worker. Returns silently if the + * index doesn't exist or has already been destroyed by the worker. + */ +void +LSNIndexAttach(void) +{ + MemoryContext oldcxt; + + if (lsn_hash != NULL) + return; /* already attached */ + + if (LSNIndexCtl == NULL || !LSNIndexCtl->is_active || + LSNIndexCtl->dsa_handle == DSA_HANDLE_INVALID) + return; + + oldcxt = MemoryContextSwitchTo(TopMemoryContext); + LWLockAcquire(LSNIndexLock, LW_SHARED); + + /* Re-check under the lock - the worker may have torn it down. */ + if (LSNIndexCtl->is_active && + LSNIndexCtl->dsa_handle != DSA_HANDLE_INVALID) + { + lsn_dsa = dsa_attach(LSNIndexCtl->dsa_handle); + dsa_pin_mapping(lsn_dsa); + lsn_hash = dshash_attach(lsn_dsa, &lsn_dsh_params, + LSNIndexCtl->hash_handle, NULL); + } + + LWLockRelease(LSNIndexLock); + MemoryContextSwitchTo(oldcxt); + + if (lsn_hash != NULL) + register_exit_callback(); +} + +void +LSNIndexDetach(void) +{ + if (lsn_hash != NULL) + { + dshash_detach(lsn_hash); + lsn_hash = NULL; + } + if (lsn_dsa != NULL) + { + dsa_detach(lsn_dsa); + lsn_dsa = NULL; + } +} + +bool +LSNIndexIsActive(void) +{ + return (LSNIndexCtl != NULL && LSNIndexCtl->is_active); +} + +/* ---------------------------------------------------------------- + * Building the index (startup process, during WAL scan) + * ---------------------------------------------------------------- + */ + +void +LSNIndexAddEntry(XLogReaderState *state) +{ + RelFileLocator rlocator; + ForkNumber forknum; + BlockNumber blkno; + BlockNumber nblocks; + SMgrRelation smgr; + BufferTag tag; + PageLSNEntry *entry; + bool found; + + Assert(lsn_dsa != NULL && lsn_hash != NULL); + + for (int blk_id = 0; blk_id <= state->record->max_block_id; blk_id++) + { + LSNNode *node; + dsa_pointer node_dp; + + if (!XLogRecHasBlockRef(state, blk_id)) + continue; + + XLogRecGetBlockTag(state, blk_id, &rlocator, &forknum, &blkno); + InitBufferTag(&tag, &rlocator, forknum, blkno); + + /* + * Make sure the page exists on disk before anyone can ask for it. A + * record can be for a block past the end of its file: the page was + * created in shared buffers and never written before the crash, or + * the file was truncated earlier in this WAL and the block recreated + * after. Stock recovery extends the file when it replays such a + * record (see XLogReadBufferExtended()); do the same now, while no + * other process can be extending the relation. Otherwise a backend + * would later be handed this block as a brand new page, and its + * pending records would be lost with it. The fork may not exist yet + * at all, like a visibility map before its first page. + */ + smgr = smgropen(rlocator, INVALID_PROC_NUMBER); + smgrcreate(smgr, forknum, true); + nblocks = smgrnblocks(smgr, forknum); + if (blkno >= nblocks) + { + smgrzeroextend(smgr, forknum, nblocks, blkno - nblocks + 1, false); + scan_blocks_extended += blkno - nblocks + 1; + } + + entry = (PageLSNEntry *) dshash_find_or_insert(lsn_hash, &tag, &found); + + if (!found) + { + entry->lsn_head = InvalidDsaPointer; + entry->lsn_tail = InvalidDsaPointer; + entry->max_lsn = InvalidXLogRecPtr; + scan_pages_indexed++; + } + + /* Allocate a new LSNNode in DSA memory */ + node_dp = dsa_allocate(lsn_dsa, sizeof(LSNNode)); + node = (LSNNode *) dsa_get_address(lsn_dsa, node_dp); + node->lsn = state->ReadRecPtr; + node->next = InvalidDsaPointer; + + /* Append to tail of linked list */ + if (!DsaPointerIsValid(entry->lsn_tail)) + { + entry->lsn_head = node_dp; + entry->lsn_tail = node_dp; + } + else + { + LSNNode *tail = (LSNNode *) dsa_get_address(lsn_dsa, entry->lsn_tail); + + tail->next = node_dp; + entry->lsn_tail = node_dp; + } + + /* Track max LSN for quick skip checks */ + if (!found || state->ReadRecPtr > entry->max_lsn) + entry->max_lsn = state->ReadRecPtr; + + dshash_release_lock(lsn_hash, entry); + } + + scan_records_indexed++; +} + +/* ---------------------------------------------------------------- + * Forgetting pages that WAL replay removes + * ---------------------------------------------------------------- + * + * While the startup process builds the index, replaying a record that drops + * a relation, truncates one, or drops a database removes pages whose pending + * records must never be replayed. Reading a dropped relation's pages would + * fail, because its files are gone; a truncated page could be created again + * later, at LSN 0, and a stale pre-truncation record would then be applied + * to it. + * + * Only entries indexed so far are removed. WAL is scanned in LSN order, so + * these are exactly the records written before the drop or truncation; + * should the relfilenumber or the block be used again later in the WAL, the + * new entries are added after this and kept. + * + * Each call scans the whole index, which is fine for a handful of drops. A + * workload dropping many relations would want a per-relation list of the + * blocks in the index. + * + * A truncation also clears the tail of the last surviving FSM and VM page. + * To do so it reads that page through the buffer manager, which replays its + * pending records first, up to the truncation, and removes its entry. The + * page then sits in shared buffers until a later record for it is indexed; + * see LSNIndexPrepareToDefer() for what happens then. + */ + +/* Free a page's list of LSNs. */ +static void +free_lsn_list(dsa_pointer node_dp) +{ + while (DsaPointerIsValid(node_dp)) + { + dsa_pointer next_dp; + + next_dp = ((LSNNode *) dsa_get_address(lsn_dsa, node_dp))->next; + dsa_free(lsn_dsa, node_dp); + node_dp = next_dp; + } +} + +/* + * LSNIndexForgetRelation - forget the pages of a relation fork from block + * 'minblkno' on. + * + * Called from XLogDropRelation() with 0 and from XLogTruncateRelation() with + * the new length, next to forgetting the fork's invalid pages (compare + * forget_invalid_pages() in xlogutils.c). + */ +void +LSNIndexForgetRelation(RelFileLocator rlocator, ForkNumber forknum, + BlockNumber minblkno) +{ + dshash_seq_status seq; + PageLSNEntry *entry; + int nforgotten = 0; + + /* Only while the index is being built, by the process building it. */ + if (lsn_hash == NULL || !InRecovery) + return; + + dshash_seq_init(&seq, lsn_hash, true); + while ((entry = (PageLSNEntry *) dshash_seq_next(&seq)) != NULL) + { + RelFileLocator tag_rlocator = BufTagGetRelFileLocator(&entry->tag); + + if (RelFileLocatorEquals(tag_rlocator, rlocator) && + BufTagGetForkNum(&entry->tag) == forknum && + entry->tag.blockNum >= minblkno) + { + free_lsn_list(entry->lsn_head); + dshash_delete_current(&seq); + nforgotten++; + } + } + dshash_seq_term(&seq); + + if (nforgotten > 0) + elog(DEBUG1, "fast recovery: forgot %d pages of relation %u/%u/%u fork %d from block %u", + nforgotten, rlocator.spcOid, rlocator.dbOid, rlocator.relNumber, + forknum, minblkno); +} + +/* + * LSNIndexForgetDatabase - forget every page of a database. + * + * Called from XLogDropDatabase() (compare forget_invalid_pages_db()). + */ +void +LSNIndexForgetDatabase(Oid dbid) +{ + dshash_seq_status seq; + PageLSNEntry *entry; + int nforgotten = 0; + + if (lsn_hash == NULL || !InRecovery) + return; + + dshash_seq_init(&seq, lsn_hash, true); + while ((entry = (PageLSNEntry *) dshash_seq_next(&seq)) != NULL) + { + if (entry->tag.dbOid == dbid) + { + free_lsn_list(entry->lsn_head); + dshash_delete_current(&seq); + nforgotten++; + } + } + dshash_seq_term(&seq); + + if (nforgotten > 0) + elog(DEBUG1, "fast recovery: forgot %d pages of dropped database %u", + nforgotten, dbid); +} + +/* + * LSNIndexPrepareToDefer - may the startup process index this record instead + * of replaying it? Makes it so where needed. + * + * A record for a page that is in shared buffers cannot just be indexed. A + * page is replayed on demand when it is read from disk, and a resident page + * is not read again while it stays in the pool, so the record would never be + * applied and the page would be handed out stale. During the scan a page + * gets into the pool only when an eagerly replayed record reads it: a + * truncation reads the last surviving FSM and VM page to clear their tails. + * That read replayed the page's indexed records up to the truncation and + * removed its entry. + * + * So write such a page out, drop it from the pool, and index the record like + * any other: the next read of the page is a miss and replays it on demand. + * The write is needed in any case, since the tail clear has no record of its + * own. An earlier version kept the page current instead, by replaying every + * later record for it at once. But replaying a record reads its other pages + * into the pool, which then had to be kept current the same way, and so on: + * one truncation of a table that fits in one VM page (about 255 MB of heap) + * made the rest of that table's records eager. + * + * EvictUnpinnedBuffer() is documented as racy because the buffer may be + * reassigned between the lookup and the eviction; that takes a concurrent + * allocation, and during the scan no other process allocates buffers. The + * background writer may hold a transient pin while writing the page, in + * which case the eviction fails and the record is replayed now, as before. + * + * Records for the init fork of an unlogged relation are always replayed now: + * at the end of recovery the init fork is copied to the main fork without + * going through shared buffers, so it must be on disk by then. + */ +bool +LSNIndexPrepareToDefer(XLogReaderState *record) +{ + for (int block_id = 0; block_id <= record->record->max_block_id; block_id++) + { + RelFileLocator rlocator; + ForkNumber forknum; + BlockNumber blkno; + BufferTag tag; + uint32 hash; + LWLock *partitionLock; + int buf_id; + bool flushed; + + if (!XLogRecHasBlockRef(record, block_id)) + continue; + + XLogRecGetBlockTag(record, block_id, &rlocator, &forknum, &blkno); + if (forknum == INIT_FORKNUM) + { + scan_records_replayed++; + return false; + } + + InitBufferTag(&tag, &rlocator, forknum, blkno); + hash = BufTableHashCode(&tag); + partitionLock = BufMappingPartitionLock(hash); + LWLockAcquire(partitionLock, LW_SHARED); + buf_id = BufTableLookup(&tag, hash); + LWLockRelease(partitionLock); + + if (buf_id < 0) + continue; + + if (!EvictUnpinnedBuffer(BufferDescriptorGetBuffer(GetBufferDescriptor(buf_id)), + &flushed)) + { + scan_records_replayed++; + return false; + } + + scan_pages_evicted++; + elog(DEBUG1, "fast recovery: evicted block %u of relation %u/%u/%u fork %d to index its records", + blkno, rlocator.spcOid, rlocator.dbOid, rlocator.relNumber, forknum); + } + + return true; +} + +/* + * LSNIndexLogScanSummary - report what the scan did, once it is over. + */ +void +LSNIndexLogScanSummary(void) +{ + if (lsn_dsa == NULL) + return; + + ereport(LOG, + errmsg("fast crash recovery: deferred %lld records for %lld pages (%zu MB of index), extended files by %lld blocks; replayed %lld records with block references immediately, evicting %lld pages", + (long long) scan_records_indexed, + (long long) scan_pages_indexed, + dsa_get_total_size(lsn_dsa) / (1024 * 1024), + (long long) scan_blocks_extended, + (long long) scan_records_replayed, + (long long) scan_pages_evicted)); +} + +/* ---------------------------------------------------------------- + * On-demand WAL replay for a single page + * ---------------------------------------------------------------- + */ + +/* + * LSNIndexScratchBuffer - a throwaway buffer for one of the current record's + * other pages. + * + * While a page is replayed on demand, the record's other pages are skipped: + * their own replay applies the record to them. Redo routines that + * initialize such a page still need a buffer to write into, so give them a + * local one, and drop it as soon as the record is done, before it could ever + * be written out under the other page's identity. + */ +Buffer +LSNIndexScratchBuffer(RelFileLocator rlocator, ForkNumber forknum, + BlockNumber blkno) +{ + SMgrRelation smgr = smgropen(rlocator, INVALID_PROC_NUMBER); + BufferDesc *bufHdr; + bool found; + + if (nscratch >= lengthof(scratch_tags)) + elog(ERROR, "too many scratch buffers for one WAL record"); + + bufHdr = LocalBufferAlloc(smgr, forknum, blkno, &found); + InitBufferTag(&scratch_tags[nscratch++], &rlocator, forknum, blkno); + + return BufferDescriptorGetBuffer(bufHdr); +} + +/* Drop the scratch buffers handed out so far. */ +static void +drop_scratch_buffers(void) +{ + for (int i = 0; i < nscratch; i++) + { + RelFileLocator rlocator = BufTagGetRelFileLocator(&scratch_tags[i]); + ForkNumber forknum = BufTagGetForkNum(&scratch_tags[i]); + BlockNumber blkno = scratch_tags[i].blockNum; + + DropRelationLocalBuffers(rlocator, &forknum, 1, &blkno); + } + nscratch = 0; +} + +/* + * Error context for on-demand replay, like rm_redo_error_callback() in normal + * recovery: name the record, and the page being recovered. + */ +static void +ondemand_redo_error_callback(void *arg) +{ + XLogReaderState *record = (XLogReaderState *) arg; + StringInfoData buf; + + initStringInfo(&buf); + xlog_outdesc(&buf, record); + + errcontext("on-demand WAL redo at %X/%08X for %s, recovering block %u of relation %s", + LSN_FORMAT_ARGS(record->ReadRecPtr), buf.data, + targetTag.blockNum, + relpathperm(BufTagGetRelFileLocator(&targetTag), + BufTagGetForkNum(&targetTag)).str); + + pfree(buf.data); +} + +/* + * Replay, in LSN order, the indexed records that touch the page 'tag'. + */ +static void +ReplayPageRecords(BufferTag *tag, dsa_pointer head_dp) +{ + dsa_pointer node_dp; + XLogReaderState *xlogreader; + char *errormsg = NULL; + ErrorContextCallback errcallback; + + xlogreader = + XLogReaderAllocate(wal_segment_size, NULL, + XL_ROUTINE(.page_read = &read_local_xlog_page, + .segment_open = wal_segment_open, + .segment_close = wal_segment_close), + NULL); + if (!xlogreader) + ereport(ERROR, + (errcode(ERRCODE_OUT_OF_MEMORY), + errmsg("out of memory"), + errdetail("Failed while allocating a WAL reading processor."))); + + for (node_dp = head_dp; + DsaPointerIsValid(node_dp); + ) + { + LSNNode *node = (LSNNode *) dsa_get_address(lsn_dsa, node_dp); + XLogRecPtr lsn = node->lsn; + dsa_pointer next_dp = node->next; + + XLogBeginRead(xlogreader, lsn); + if (!XLogReadRecord(xlogreader, &errormsg)) + { + if (errormsg) + ereport(ERROR, + (errcode_for_file_access(), + errmsg("could not read WAL at %X/%X: %s", + LSN_FORMAT_ARGS(lsn), errormsg))); + else + ereport(ERROR, + (errcode_for_file_access(), + errmsg("could not read WAL at %X/%X", + LSN_FORMAT_ARGS(lsn)))); + } + + /* Find the block reference matching our target tag and replay */ + for (int blk_id = 0; blk_id <= xlogreader->record->max_block_id; blk_id++) + { + RelFileLocator rlocator; + ForkNumber forknum; + BlockNumber blkno; + BufferTag wal_tag; + + if (!XLogRecHasBlockRef(xlogreader, blk_id)) + continue; + + XLogRecGetBlockTag(xlogreader, blk_id, &rlocator, &forknum, &blkno); + InitBufferTag(&wal_tag, &rlocator, forknum, blkno); + + if (BufferTagsEqual(tag, &wal_tag)) + { + errcallback.callback = ondemand_redo_error_callback; + errcallback.arg = (void *) xlogreader; + errcallback.previous = error_context_stack; + error_context_stack = &errcallback; + + GetRmgr(XLogRecGetRmid(xlogreader)).rm_redo(xlogreader); + + error_context_stack = errcallback.previous; + + /* the record is in; its other pages' scratch buffers can go */ + drop_scratch_buffers(); + break; + } + } + + node_dp = next_dp; + } + + XLogReaderFree(xlogreader); +} + +/* + * Start rmgr-specific recovery state for on-demand replay, once per process. + * + * The startup process has its own, set up by PerformWalRecovery(), and must + * not call RmgrStartup() again: gin, gist and spgist keep a static memory + * context that their startup creates and their cleanup deletes, so a nested + * call would pull it out from under the scan. Anywhere else, start them + * once, in TopMemoryContext so that the contexts outlive the query that + * happened to read the first page, and keep them until the process exits. + */ +static void +StartRmgrsForOnDemandReplay(void) +{ + MemoryContext oldcxt; + + if (rmgrs_started || AmStartupProcess()) + return; + + oldcxt = MemoryContextSwitchTo(TopMemoryContext); + RmgrStartup(); + MemoryContextSwitchTo(oldcxt); + rmgrs_started = true; +} + +/* + * LSNIndexBeginPageReplay - does this page have pending records? If so, + * register us as replaying it. + * + * Called by the buffer manager for a page it's about to read from disk. On + * false, there's nothing to replay: read the page as usual. On true, the + * caller must read the page, replay it, mark it valid, and call + * LSNIndexEndPageReplay(); if any of that fails, LSNIndexErrorCleanup() + * undoes it instead. Until then FastRecoveryInProgress() counts us, so no + * checkpoint can start before the page is valid and dirty. + * + * is_active is checked and the counter bumped under LSNIndexLock, so + * LSNIndexFinish(), which clears is_active under the same lock in exclusive + * mode, sees every replay that got in. + * + * Replays never nest. Redo of an indexed page reaches that page through + * XLogReadBufferExtended(), which hands it the buffer being replayed into; + * the record's other pages are skipped by XLogReadBufferForRedoExtended(), + * and the free space map is left alone by XLogRecordPageWithFreeSpace(). So + * nothing redo does can bring us back here, and a read that does is a bug: it + * would wait forever for the I/O we hold if it's the same page, or replay + * another page inside this one's redo if not. (The startup process can get + * here from its own, eager redo of a record that touches an indexed page; + * that's one level, not nesting, since the replay it starts reads nothing.) + */ +bool +LSNIndexBeginPageReplay(const BufferTag *tag) +{ + PageLSNEntry *entry; + bool pending = false; + + if (LSNIndexCtl == NULL || !LSNIndexCtl->is_active) + return false; + + /* Attach before taking LSNIndexLock: LSNIndexAttach() takes it too. */ + if (lsn_hash == NULL) + LSNIndexAttach(); + if (lsn_hash == NULL) + return false; + + if (replay_begun) + elog(ERROR, "block %u of relation %s read during on-demand replay of block %u of relation %s", + tag->blockNum, + relpathperm(BufTagGetRelFileLocator(tag), + BufTagGetForkNum(tag)).str, + replaying_tag.blockNum, + relpathperm(BufTagGetRelFileLocator(&replaying_tag), + BufTagGetForkNum(&replaying_tag)).str); + + LWLockAcquire(LSNIndexLock, LW_SHARED); + if (LSNIndexCtl->is_active) + { + entry = (PageLSNEntry *) dshash_find(lsn_hash, tag, false); + if (entry != NULL) + { + dshash_release_lock(lsn_hash, entry); + pg_atomic_fetch_add_u32(&LSNIndexCtl->nreplaying, 1); + pending = true; + } + } + LWLockRelease(LSNIndexLock); + + if (pending) + { + replay_begun = true; + replaying_tag = *tag; + } + + return pending; +} + +/* + * LSNIndexEndPageReplay - undo LSNIndexBeginPageReplay(). + */ +void +LSNIndexEndPageReplay(void) +{ + Assert(replay_begun); + replay_begun = false; + pg_atomic_fetch_sub_u32(&LSNIndexCtl->nreplaying, 1); +} + +/* + * LSNIndexErrorCleanup - undo an on-demand replay that an error cut short. + * + * Called from AbortTransaction() and AbortSubTransaction(), next to + * pgaio_error_cleanup() and before the resource owner that started the + * buffer's I/O ends it, and at process exit. Redo dirties the buffer before + * it is valid, which nothing else ever does, and AbortBufferIO() rightly + * insists that an invalid buffer is clean; so clean it here, and let the + * resource owner end the I/O as for any failed read. The page was never + * valid and its entry is still in the index, so the next reader starts over. + * Then forget the replay, so that this process can read pages again and + * stops holding checkpoints off. + * + * Scratch buffers handed to redo may still be pinned at this point; the next + * replay drops them, once the pins are gone. + */ +void +LSNIndexErrorCleanup(void) +{ + if (!replay_begun) + return; + + if (inReplayPageWals) + { + AbortPendingBufferReplay(targetBuffer); + inReplayPageWals = false; + targetBuffer = InvalidBuffer; + InRecovery = saved_in_recovery; + } + + LSNIndexEndPageReplay(); +} + +/* + * LSNIndexReplayIntoBuffer - apply a page's pending records to its buffer. + * + * The caller holds 'buffer' pinned and its I/O-in-progress flag, and has + * just read the page from disk into it. Nobody else can see the page until + * the caller marks it valid, so redo needs no lock to keep readers out, and + * nobody can be holding a pointer into the page. Redo reaches this page + * through XLogReadBufferExtended(), which hands it this buffer instead of + * reading the page again; the record's other pages are skipped, because + * their own replay applies the record to them. + */ +void +LSNIndexReplayIntoBuffer(Buffer buffer) +{ + BufferTag tag = GetBufferDescriptor(buffer - 1)->tag; + PageLSNEntry *entry; + dsa_pointer head_dp; + + Assert(replay_begun && BufferTagsEqual(&replaying_tag, &tag)); + Assert(!inReplayPageWals); + + /* + * Scratch buffers left over by a replay that failed: their pins are gone + * with the aborted transaction, so they can go now. + */ + if (nscratch > 0) + drop_scratch_buffers(); + + /* Only the process holding the page's I/O retires its entry: us. */ + entry = (PageLSNEntry *) dshash_find(lsn_hash, &tag, false); + if (entry == NULL) + elog(ERROR, "no pending WAL records for block %u of relation %s", + tag.blockNum, + relpathperm(BufTagGetRelFileLocator(&tag), + BufTagGetForkNum(&tag)).str); + head_dp = entry->lsn_head; + dshash_release_lock(lsn_hash, entry); + + StartRmgrsForOnDemandReplay(); + + /* + * Redo routines were written to run in the startup process and use + * InRecovery to mean "applying WAL, not generating it": + * visibilitymap_set() asserts it, since outside recovery setting a bit + * must happen in the critical section that logs it; mdreadv() zero-fills + * a short read under it, as recovery of a relation the OS crash left + * short requires. On-demand replay is that same work done later, in + * another process, so say so for the duration of the redo calls. Nothing + * redo does under closure consults the flag for its other meaning, "I am + * the startup process": it reads no page but the one it's handed. + * + * Should redo fail, LSNIndexErrorCleanup() puts all of this back. + */ + saved_in_recovery = InRecovery; + InRecovery = true; + inReplayPageWals = true; + targetTag = tag; + targetBuffer = buffer; + + ReplayPageRecords(&tag, head_dp); + + inReplayPageWals = false; + targetBuffer = InvalidBuffer; + InRecovery = saved_in_recovery; +} + +/* + * LSNIndexForgetPage - drop a page's pending records. + * + * Called once a replayed page has been marked valid: a valid page is + * current, and must never be replayed again (a full-page image in its list + * would roll it back). Also called when a page is about to be zeroed for + * reuse, as its old records no longer apply. The caller holds the buffer + * pinned, so the page can't be evicted and read back in before we're done. + */ +void +LSNIndexForgetPage(const BufferTag *tag) +{ + PageLSNEntry *entry; + + if (LSNIndexCtl == NULL || !LSNIndexCtl->is_active) + return; + if (lsn_hash == NULL) + LSNIndexAttach(); + if (lsn_hash == NULL) + return; + + /* is_active, seen under the lock, keeps LSNIndexFinish() from freeing it */ + LWLockAcquire(LSNIndexLock, LW_SHARED); + if (LSNIndexCtl->is_active) + { + entry = (PageLSNEntry *) dshash_find(lsn_hash, tag, true); + if (entry != NULL) + { + free_lsn_list(entry->lsn_head); + dshash_delete_entry(lsn_hash, entry); + } + } + LWLockRelease(LSNIndexLock); +} + +/* ---------------------------------------------------------------- + * Accessor for the dshash table (used by fast recovery worker) + * ---------------------------------------------------------------- + */ + +dshash_table * +LSNIndexGetHash(void) +{ + Assert(lsn_hash != NULL); + return lsn_hash; +} + +/* ---------------------------------------------------------------- + * End of fast recovery + * ---------------------------------------------------------------- + */ + +/* + * FastRecoveryInProgress - may pre-crash WAL still be replayed onto a page? + * + * True while the index is active, and afterwards until the last replay that + * was already running has finished. No checkpoint may start while this is + * true. A completed checkpoint promises that everything before its redo + * point is on disk, and a crash after it starts recovery at that redo point. + * A page that has not been replayed yet, or that a running replay dirties + * after the checkpoint has collected its dirty buffers, would then never get + * its pre-crash WAL replayed. + */ +bool +FastRecoveryInProgress(void) +{ + bool result; + + if (LSNIndexCtl == NULL) + return false; + + LWLockAcquire(LSNIndexLock, LW_SHARED); + result = LSNIndexCtl->is_active || + pg_atomic_read_u32(&LSNIndexCtl->nreplaying) > 0; + LWLockRelease(LSNIndexLock); + + return result; +} + +/* + * Free the DSA and dshash. Only safe once nobody can reach them anymore. + */ +static void +LSNIndexDestroy(void) +{ + Assert(!LSNIndexCtl->is_active); + Assert(pg_atomic_read_u32(&LSNIndexCtl->nreplaying) == 0); + + if (lsn_hash != NULL) + { + dshash_destroy(lsn_hash); + lsn_hash = NULL; + } + if (lsn_dsa != NULL) + { + dsa_unpin(lsn_dsa); + dsa_detach(lsn_dsa); + lsn_dsa = NULL; + } + + LWLockAcquire(LSNIndexLock, LW_EXCLUSIVE); + LSNIndexCtl->dsa_handle = DSA_HANDLE_INVALID; + LSNIndexCtl->hash_handle = DSHASH_HANDLE_INVALID; + LWLockRelease(LSNIndexLock); +} + +/* + * LSNIndexFinish - end fast recovery. + * + * Called by the fast recovery worker once it has read every page in the + * index. Stops new on-demand replays, waits for the ones already running, + * and only then frees the index. FastRecoveryInProgress() turns false when + * the wait is over, not before. + */ +void +LSNIndexFinish(void) +{ + LWLockAcquire(LSNIndexLock, LW_EXCLUSIVE); + LSNIndexCtl->is_active = false; + LWLockRelease(LSNIndexLock); + + while (pg_atomic_read_u32(&LSNIndexCtl->nreplaying) > 0) + { + (void) WaitLatch(MyLatch, + WL_LATCH_SET | WL_TIMEOUT | WL_EXIT_ON_PM_DEATH, + 10L, WAIT_EVENT_FAST_RECOVERY_DRAIN); + ResetLatch(MyLatch); + } + + LSNIndexDestroy(); +} diff --git a/src/backend/access/transam/meson.build b/src/backend/access/transam/meson.build index 06aadc7f315..2371237b053 100644 --- a/src/backend/access/transam/meson.build +++ b/src/backend/access/transam/meson.build @@ -4,6 +4,7 @@ backend_sources += files( 'clog.c', 'commit_ts.c', 'generic_xlog.c', + 'lsn_indexer.c', 'multixact.c', 'parallel.c', 'rmgr.c', diff --git a/src/backend/access/transam/xact.c b/src/backend/access/transam/xact.c index cab99f25e1e..c019ba2e346 100644 --- a/src/backend/access/transam/xact.c +++ b/src/backend/access/transam/xact.c @@ -21,6 +21,7 @@ #include #include "access/commit_ts.h" +#include "access/lsn_indexer.h" #include "access/multixact.h" #include "access/parallel.h" #include "access/subtrans.h" @@ -2896,6 +2897,7 @@ AbortTransaction(void) pgstat_progress_end_command(); pgaio_error_cleanup(); + LSNIndexErrorCleanup(); /* Clean up buffer content locks, too */ UnlockBuffers(); @@ -5322,6 +5324,7 @@ AbortSubTransaction(void) pgstat_progress_end_command(); pgaio_error_cleanup(); + LSNIndexErrorCleanup(); UnlockBuffers(); diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c index 63d214782da..8914ce0cced 100644 --- a/src/backend/access/transam/xlog.c +++ b/src/backend/access/transam/xlog.c @@ -109,6 +109,7 @@ #include "utils/timestamp.h" #include "utils/varlena.h" #include "utils/wait_event.h" +#include "access/lsn_indexer.h" #ifdef WAL_DEBUG #include "utils/memutils.h" @@ -6070,6 +6071,12 @@ CheckRequiredParameterValues(void) } } +int +get_control_state(void) +{ + return ControlFile->state; +} + /* * This must be called ONCE during postmaster or standalone-backend startup */ @@ -6550,7 +6557,14 @@ StartupXLOG(void) * We're all set for replaying the WAL now. Do it. */ PerformWalRecovery(); - performedWalRecovery = true; + + /* + * With fast crash recovery active, WAL has been scanned but not + * applied: the end-of-recovery checkpoint and the other work that + * follows a completed replay are for the fast recovery worker to + * request once every page has been recovered. + */ + performedWalRecovery = !LSNIndexIsActive(); } else performedWalRecovery = false; @@ -7711,6 +7725,20 @@ CreateCheckPoint(int flags) if (RecoveryInProgress() && (flags & CHECKPOINT_END_OF_RECOVERY) == 0) elog(ERROR, "can't create a checkpoint during recovery"); + /* + * A completed checkpoint promises that everything before its redo point + * is on disk. While fast crash recovery may still replay pre-crash WAL + * onto some page, that promise would be false; see + * FastRecoveryInProgress(). Erroring out here also fails any backend + * waiting for this checkpoint, via the checkpointer's error handling. + */ + if (FastRecoveryInProgress()) + ereport(ERROR, + (errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE), + errmsg("cannot create a checkpoint while fast crash recovery is in progress"), + errdetail("Some pages have not been recovered from WAL yet."), + errhint("Retry after the fast recovery worker has finished."))); + /* * Prepare to accumulate statistics. * @@ -8498,6 +8526,15 @@ CreateRestartPoint(int flags) /* Concurrent checkpoint/restartpoint cannot happen */ Assert(!IsUnderPostmaster || MyBackendType == B_CHECKPOINTER); + /* + * A restartpoint moves the redo point in pg_control just like a + * checkpoint, so it must not happen while fast crash recovery has pages + * left to replay either. Report it as not performed; the checkpointer + * then retries later. + */ + if (FastRecoveryInProgress()) + return false; + /* Get a local copy of the last safe checkpoint record. */ SpinLockAcquire(&XLogCtl->info_lck); lastCheckPointRecPtr = XLogCtl->lastCheckPointRecPtr; diff --git a/src/backend/access/transam/xlogrecovery.c b/src/backend/access/transam/xlogrecovery.c index 54aaec9529f..5cd6094c2b8 100644 --- a/src/backend/access/transam/xlogrecovery.c +++ b/src/backend/access/transam/xlogrecovery.c @@ -27,6 +27,7 @@ #include #include #include +#include #include #include @@ -68,6 +69,7 @@ #include "utils/ps_status.h" #include "utils/pg_rusage.h" #include "utils/wait_event.h" +#include "access/lsn_indexer.h" /* Unsupported old recovery command file names (relative to $PGDATA) */ #define RECOVERY_COMMAND_FILE "recovery.conf" @@ -350,6 +352,7 @@ static bool read_tablespace_map(List **tablespaces); static void xlogrecovery_redo(XLogReaderState *record, TimeLineID replayTLI); static void CheckRecoveryConsistency(void); static void rm_redo_error_callback(void *arg); +static void MaybeStartFastRecovery(void); #ifdef WAL_DEBUG static void xlog_outrec(StringInfo buf, XLogReaderState *record); #endif @@ -417,6 +420,31 @@ XLogRecoveryShmemInit(void *arg) ConditionVariableInit(&XLogRecoveryCtl->recoveryNotPausedCV); } +/* + * Start fast crash recovery, if it's enabled and applies. + * + * It applies to crash recovery under the postmaster only. Archive recovery + * and standby mode replay WAL that has no end, so the index would never be + * drained and restartpoints never allowed; single-user mode has no worker to + * drain it either. Those replay WAL the usual way. + */ +static void +MaybeStartFastRecovery(void) +{ + if (!fast_crash_recovery) + return; + + if (ArchiveRecoveryRequested || !IsUnderPostmaster) + { + ereport(LOG, + (errmsg("fast crash recovery is not used in %s; replaying WAL normally", + ArchiveRecoveryRequested ? "archive recovery or standby mode" : "single-user mode"))); + return; + } + + LSNIndexInit(); +} + /* * A thin wrapper to enable StandbyMode and do other preparatory work as * needed. @@ -876,9 +904,13 @@ InitWalRecovery(ControlFileData *ControlFile, bool *wasShutdown_ptr, ereport(PANIC, (errmsg("invalid redo record in shutdown checkpoint"))); InRecovery = true; + MaybeStartFastRecovery(); } else if (ControlFile->state != DB_SHUTDOWNED) + { InRecovery = true; + MaybeStartFastRecovery(); + } else if (ArchiveRecoveryRequested) { /* force recovery due to presence of recovery signal file */ @@ -1869,6 +1901,9 @@ PerformWalRecovery(void) (errmsg("last completed transaction was at log time %s", timestamptz_to_str(xtime)))); + if (LSNIndexIsActive()) + LSNIndexLogScanSummary(); + InRedo = false; } else @@ -1898,6 +1933,7 @@ ApplyWalRecord(XLogReaderState *xlogreader, XLogRecord *record, TimeLineID *repl { ErrorContextCallback errcallback; bool switchedTLI = false; + bool deferred = false; /* Setup error traceback support for ereport() */ errcallback.callback = rm_redo_error_callback; @@ -1970,21 +2006,42 @@ ApplyWalRecord(XLogReaderState *xlogreader, XLogRecord *record, TimeLineID *repl RecordKnownAssignedTransactionIds(record->xl_xid); /* - * Some XLOG record types that are related to recovery are processed - * directly here, rather than in xlog_redo() + * Fast crash recovery defers only the redo of relation pages. A record + * that references blocks is indexed by page and replayed when the page is + * first read. Every other record is replayed now: it reads no relation + * pages, so it is cheap, and its effect is global (transaction status, + * SLRUs, which files exist, where the catalogs live), so no later event + * could trigger its replay. A page that such a record read into shared + * buffers is evicted before a record for it is indexed; see + * LSNIndexPrepareToDefer(). */ - if (record->xl_rmid == RM_XLOG_ID) - xlogrecovery_redo(xlogreader, *replayTLI); + if (LSNIndexIsActive() && XLogRecHasAnyBlockRefs(xlogreader) && + LSNIndexPrepareToDefer(xlogreader)) + { + LSNIndexAddEntry(xlogreader); + deferred = true; + } + else + { + /* + * Some XLOG record types that are related to recovery are processed + * directly here, rather than in xlog_redo() + */ + if (record->xl_rmid == RM_XLOG_ID) + xlogrecovery_redo(xlogreader, *replayTLI); - /* Now apply the WAL record itself */ - GetRmgr(record->xl_rmid).rm_redo(xlogreader); + /* Now apply the WAL record itself */ + GetRmgr(record->xl_rmid).rm_redo(xlogreader); + } /* * After redo, check whether the backup pages associated with the WAL * record are consistent with the existing pages. This check is done only - * if consistency check is enabled for this record. + * if consistency check is enabled for this record, and only for records + * applied here: a deferred record's pages are replayed later, by whoever + * reads them, and reading them now would replay them past this record. */ - if ((record->xl_info & XLR_CHECK_CONSISTENCY) != 0) + if (!deferred && (record->xl_info & XLR_CHECK_CONSISTENCY) != 0) verifyBackupPageConsistency(xlogreader); /* Pop the error context stack */ @@ -4270,6 +4327,18 @@ XLogFileRead(XLogSegNo segno, TimeLineID tli, /* Success! */ curFileTLI = tli; +#if defined(USE_POSIX_FADVISE) && defined(POSIX_FADV_WILLNEED) + + /* + * Fast crash recovery decodes WAL faster than the kernel grows its + * readahead window for 8 kB reads, and then waits at nearly every + * window. Ask for the whole segment at once; the kernel reads it in + * the background while we decode what has already arrived. + */ + if (LSNIndexIsActive()) + (void) posix_fadvise(fd, 0, 0, POSIX_FADV_WILLNEED); +#endif + /* Report recovery progress in PS display */ snprintf(activitymsg, sizeof(activitymsg), "recovering %s", xlogfname); diff --git a/src/backend/access/transam/xlogutils.c b/src/backend/access/transam/xlogutils.c index ff71e14015d..857eb23a4c6 100644 --- a/src/backend/access/transam/xlogutils.c +++ b/src/backend/access/transam/xlogutils.c @@ -28,6 +28,7 @@ #include "storage/smgr.h" #include "utils/hsearch.h" #include "utils/rel.h" +#include "access/lsn_indexer.h" /* GUC variable */ @@ -36,8 +37,10 @@ bool ignore_invalid_pages = false; /* * Are we doing recovery from XLOG? * - * This is only ever true in the startup process; it should be read as meaning - * "this process is replaying WAL records", rather than "the system is in + * This is true in the startup process during recovery, and in any process + * while it replays a page's WAL on demand after a fast crash recovery (see + * LSNIndexReplayIntoBuffer()); it should be read as meaning "this process is + * replaying WAL records", rather than "the system is in * recovery mode". It should be examined primarily by functions that need * to act differently when called from a WAL redo function (e.g., to skip WAL * logging). To check whether the system is in recovery regardless of which @@ -381,6 +384,44 @@ XLogReadBufferForRedoExtended(XLogReaderState *record, block_id); } + /* + * During on-demand replay of one page, skip the record's other pages: + * their own replay applies the record to them. Callers that initialize a + * page use the buffer whatever we return, so they get a scratch one. A + * block with an image to apply is always BLK_RESTORED in normal recovery, + * and some redo routines rely on that (the FPI loop in xlog_redo(), hash + * and gin page initialization), so restore the image into a scratch + * buffer for them; the page's own replay restores it for real. + */ + if (inReplayPageWals) + { + BufferTag recordTag; + + InitBufferTag(&recordTag, &rlocator, forknum, blkno); + + if (!BufferTagsEqual(&targetTag, &recordTag)) + { + if (XLogRecBlockImageApply(record, block_id)) + { + *buf = LSNIndexScratchBuffer(rlocator, forknum, blkno); + page = BufferGetPage(*buf); + if (!RestoreBlockImage(record, block_id, page)) + ereport(ERROR, + (errcode(ERRCODE_INTERNAL_ERROR), + errmsg_internal("%s", record->errormsg_buf))); + if (!PageIsNew(page)) + PageSetLSN(page, lsn); + MarkBufferDirty(*buf); + return BLK_RESTORED; + } + if (mode == RBM_ZERO_AND_LOCK || mode == RBM_ZERO_AND_CLEANUP_LOCK) + *buf = LSNIndexScratchBuffer(rlocator, forknum, blkno); + else + *buf = InvalidBuffer; + return BLK_DONE; + } + } + /* * Make sure that if the block is marked with WILL_INIT, the caller is * going to initialize it. And vice versa. @@ -436,7 +477,15 @@ XLogReadBufferForRedoExtended(XLogReaderState *record, { if (mode != RBM_ZERO_AND_LOCK && mode != RBM_ZERO_AND_CLEANUP_LOCK) { - if (get_cleanup_lock) + /* + * A page being replayed on demand isn't valid yet, so no one + * else can be looking at it, and an exclusive lock is as good + * as a cleanup lock (cf. ZeroAndLockBuffer()). A real + * cleanup lock would also wait for the pins of backends that + * are waiting for this very page's I/O to finish. + */ + if (get_cleanup_lock && + !(inReplayPageWals && *buf == targetBuffer)) LockBufferForCleanup(*buf); else LockBuffer(*buf, BUFFER_LOCK_EXCLUSIVE); @@ -491,6 +540,31 @@ XLogReadBufferExtended(RelFileLocator rlocator, ForkNumber forknum, Assert(blkno != P_NEW); + /* + * During on-demand replay of a page, redo reaches that page here. Hand + * it the buffer the page is being replayed into, rather than reading the + * page again, which would wait for the I/O our caller holds. Pin it once + * more, because redo releases what it gets. For the zeroing modes, zero + * and lock it as the buffer manager would; an exclusive lock serves for + * RBM_ZERO_AND_CLEANUP_LOCK too, because no one else can see the page. + */ + if (inReplayPageWals) + { + BufferTag tag; + + InitBufferTag(&tag, &rlocator, forknum, blkno); + if (BufferTagsEqual(&tag, &targetTag)) + { + IncrBufferRefCount(targetBuffer); + if (mode == RBM_ZERO_AND_LOCK || mode == RBM_ZERO_AND_CLEANUP_LOCK) + { + memset(BufferGetPage(targetBuffer), 0, BLCKSZ); + LockBuffer(targetBuffer, BUFFER_LOCK_EXCLUSIVE); + } + return targetBuffer; + } + } + /* Do we have a clue where the buffer might be already? */ if (BufferIsValid(recent_buffer) && mode == RBM_NORMAL && @@ -544,7 +618,7 @@ XLogReadBufferExtended(RelFileLocator rlocator, ForkNumber forknum, } recent_buffer_fast_path: - if (mode == RBM_NORMAL) + if (mode == RBM_NORMAL && !inReplayPageWals) { /* check that page has been initialized */ Page page = BufferGetPage(buffer); @@ -648,12 +722,14 @@ FreeFakeRelcacheEntry(Relation fakerel) * Drop a relation during XLOG replay * * This is called when the relation is about to be deleted; we need to remove - * any open "invalid-page" records for the relation. + * any open "invalid-page" records for the relation, and during fast crash + * recovery its pending records in the LSN index. */ void XLogDropRelation(RelFileLocator rlocator, ForkNumber forknum) { forget_invalid_pages(rlocator, forknum, 0); + LSNIndexForgetRelation(rlocator, forknum, 0); } /* @@ -673,6 +749,7 @@ XLogDropDatabase(Oid dbid) smgrdestroyall(); forget_invalid_pages_db(dbid); + LSNIndexForgetDatabase(dbid); } /* @@ -685,6 +762,7 @@ XLogTruncateRelation(RelFileLocator rlocator, ForkNumber forkNum, BlockNumber nblocks) { forget_invalid_pages(rlocator, forkNum, nblocks); + LSNIndexForgetRelation(rlocator, forkNum, nblocks); } /* diff --git a/src/backend/catalog/storage.c b/src/backend/catalog/storage.c index e443a4993c5..c77425e7f43 100644 --- a/src/backend/catalog/storage.c +++ b/src/backend/catalog/storage.c @@ -19,6 +19,7 @@ #include "postgres.h" +#include "access/lsn_indexer.h" #include "access/visibilitymap.h" #include "access/xact.h" #include "access/xlog.h" @@ -1055,6 +1056,8 @@ smgr_redo(XLogReaderState *record) { forks[nforks] = FSM_FORKNUM; old_blocks[nforks] = smgrnblocks(reln, FSM_FORKNUM); + /* fast crash recovery: forget the FSM pages being removed */ + LSNIndexForgetRelation(xlrec->rlocator, FSM_FORKNUM, blocks[nforks]); nforks++; need_fsm_vacuum = true; } @@ -1067,6 +1070,9 @@ smgr_redo(XLogReaderState *record) { forks[nforks] = VISIBILITYMAP_FORKNUM; old_blocks[nforks] = smgrnblocks(reln, VISIBILITYMAP_FORKNUM); + /* fast crash recovery: forget the VM pages being removed */ + LSNIndexForgetRelation(xlrec->rlocator, VISIBILITYMAP_FORKNUM, + blocks[nforks]); nforks++; } } diff --git a/src/backend/postmaster/Makefile b/src/backend/postmaster/Makefile index 55044b2bc6f..3f101394bd2 100644 --- a/src/backend/postmaster/Makefile +++ b/src/backend/postmaster/Makefile @@ -28,6 +28,7 @@ OBJS = \ startup.o \ syslogger.o \ walsummarizer.o \ - walwriter.o + walwriter.o \ + fast_recovery_worker.o include $(top_srcdir)/src/backend/common.mk diff --git a/src/backend/postmaster/checkpointer.c b/src/backend/postmaster/checkpointer.c index 580c7944119..a99fd53d031 100644 --- a/src/backend/postmaster/checkpointer.c +++ b/src/backend/postmaster/checkpointer.c @@ -39,6 +39,7 @@ #include #include +#include "access/lsn_indexer.h" #include "access/xlog.h" #include "access/xlog_internal.h" #include "access/xlogrecovery.h" @@ -183,6 +184,13 @@ static double ckpt_cached_elapsed; static pg_time_t last_checkpoint_time; static pg_time_t last_xlog_switch_time; +/* + * Flags of checkpoint requests postponed while fast crash recovery was in + * progress, and whether we have said so in the log yet. + */ +static int postponed_ckpt_flags = 0; +static bool postponed_ckpt_logged = false; + /* Prototypes for private functions */ static void ProcessCheckpointerInterrupts(void); @@ -412,6 +420,49 @@ CheckpointerMain(const void *startup_data, size_t startup_data_len) flags |= CHECKPOINT_CAUSE_TIME; } + /* + * No checkpoint or restartpoint may complete while fast crash + * recovery may still replay pre-crash WAL onto some page; see + * FastRecoveryInProgress(). A request nobody waits for is postponed: + * we move it out of shared memory, so that the loop below doesn't + * spin on it, and remember it for when fast recovery is over. A + * request somebody waits for goes ahead, and CreateCheckPoint() fails + * it with an error, which our error handling reports to the waiter. + */ + if (do_checkpoint && FastRecoveryInProgress()) + { + bool waited; + + SpinLockAcquire(&CheckpointerShmem->ckpt_lck); + waited = (CheckpointerShmem->ckpt_flags & CHECKPOINT_WAIT) != 0; + if (!waited) + { + postponed_ckpt_flags |= CheckpointerShmem->ckpt_flags | flags; + CheckpointerShmem->ckpt_flags = 0; + } + SpinLockRelease(&CheckpointerShmem->ckpt_lck); + + if (!waited) + { + if (!postponed_ckpt_logged) + { + ereport(LOG, + (errmsg("checkpoint postponed until fast crash recovery completes"))); + postponed_ckpt_logged = true; + } + do_checkpoint = false; + /* next time-driven attempt is one checkpoint_timeout away */ + last_checkpoint_time = now; + } + } + else if (!do_checkpoint && postponed_ckpt_flags != 0 && + !FastRecoveryInProgress()) + { + /* fast crash recovery is over; run what we postponed */ + do_checkpoint = true; + chkpt_or_rstpt_requested = true; + } + /* * Do a checkpoint if requested. */ @@ -434,6 +485,9 @@ CheckpointerMain(const void *startup_data, size_t startup_data_len) CheckpointerShmem->ckpt_started++; SpinLockRelease(&CheckpointerShmem->ckpt_lck); + /* include requests postponed during fast crash recovery */ + flags |= postponed_ckpt_flags; + ConditionVariableBroadcast(&CheckpointerShmem->start_cv); /* @@ -498,6 +552,13 @@ CheckpointerMain(const void *startup_data, size_t startup_data_len) else ckpt_performed = CreateRestartPoint(flags); + /* + * The postponed requests are covered now. (If the checkpoint + * failed with an error, we never get here and keep them.) + */ + postponed_ckpt_flags = 0; + postponed_ckpt_logged = false; + /* * After any checkpoint, free all smgr objects. Otherwise we * would never do so for dropped relations, as the checkpointer diff --git a/src/backend/postmaster/fast_recovery_worker.c b/src/backend/postmaster/fast_recovery_worker.c new file mode 100644 index 00000000000..7b4f360751b --- /dev/null +++ b/src/backend/postmaster/fast_recovery_worker.c @@ -0,0 +1,143 @@ +/*------------------------------------------------------------------------- + * + * fast_recovery_worker.c + * Auxiliary process that recovers the pages left pending by fast crash + * recovery, so that none is left for a backend to replay on first use + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * + * IDENTIFICATION + * src/backend/postmaster/fast_recovery_worker.c + * + *------------------------------------------------------------------------- + */ +#include "postgres.h" + +#include "access/lsn_indexer.h" +#include "access/rmgr.h" +#include "access/xlog.h" +#include "access/xlog_internal.h" +#include "access/xlogreader.h" +#include "access/xlogutils.h" +#include "lib/dshash.h" +#include "libpq/pqsignal.h" +#include "miscadmin.h" +#include "postmaster/auxprocess.h" +#include "postmaster/bgwriter.h" +#include "postmaster/fast_recovery_worker.h" +#include "postmaster/interrupt.h" +#include "storage/bufmgr.h" +#include "storage/ipc.h" +#include "storage/procsignal.h" + +/* + * FastRecoveryMain - read every page in the LSN index, which replays it. + * + * dshash_seq_next() holds a partition lock until the next call, and reading + * a page runs on-demand replay, which looks the page up in the same table; + * LWLocks are not reentrant. So collect the tags first, end the scan, and + * only then read the pages. + */ +static void +FastRecoveryMain(void) +{ + dshash_seq_status seq; + PageLSNEntry *entry; + BufferTag *tags = NULL; + int ntags = 0; + int ncap = 0; + + ereport(LOG, + errmsg("fast recovery worker started")); + + LSNIndexAttach(); + + dshash_seq_init(&seq, LSNIndexGetHash(), false); + while ((entry = (PageLSNEntry *) dshash_seq_next(&seq)) != NULL) + { + if (ntags == ncap) + { + ncap = ncap ? ncap * 2 : 1024; + if (tags == NULL) + tags = (BufferTag *) palloc(sizeof(BufferTag) * ncap); + else + tags = (BufferTag *) repalloc(tags, sizeof(BufferTag) * ncap); + } + tags[ntags++] = entry->tag; + } + dshash_seq_term(&seq); + + ereport(LOG, + errmsg_plural("fast recovery worker replaying %d pending page", + "fast recovery worker replaying %d pending pages", + ntags, ntags)); + + /* Reading a page replays its pending records; see bufmgr.c. */ + for (int i = 0; i < ntags; i++) + { + Buffer buffer; + + /* keep up with ProcSignal barriers and config reloads */ + ProcessMainLoopInterrupts(); + + buffer = ReadBufferWithoutRelcache(BufTagGetRelFileLocator(&tags[i]), + tags[i].forkNum, + tags[i].blockNum, + RBM_NORMAL, NULL, true); + ReleaseBuffer(buffer); + } + + if (tags) + pfree(tags); + + /* + * Every page has been replayed. End fast recovery: stop new replays, + * wait out the running ones, free the index. Checkpoints are allowed + * from then on. + */ + LSNIndexFinish(); + + /* + * Checkpoints were refused until now, so the last one is still the one + * from before the crash, and another crash would have to recover all of + * that WAL again. Ask for a new checkpoint, like the end-of-recovery + * checkpoint normal crash recovery takes, but without waiting for it. + */ + RequestCheckpoint(CHECKPOINT_FORCE); + + ereport(LOG, + errmsg("fast crash recovery complete")); +} + +void +FastRecoveryWorkerMain(const void *startup_data, size_t startup_data_len) +{ + MyBackendType = B_FAST_RECOVERY_WORKER; + AuxiliaryProcessMainCommon(); + + /* + * Properly accept or ignore signals that might be sent to us. + * + * SIGTERM is ignored on purpose. A smart or fast shutdown must not write + * its shutdown checkpoint while pages are still unrecovered, so the + * postmaster waits for us to finish instead of stopping us: we are in the + * set of processes it waits for in PM_WAIT_BACKENDS. An immediate + * shutdown uses SIGQUIT, whose handler InitPostmasterChild already set + * up; it writes no checkpoint, so the next startup recovers again. + */ + pqsignal(SIGHUP, SignalHandlerForConfigReload); + pqsignal(SIGINT, PG_SIG_IGN); + pqsignal(SIGTERM, PG_SIG_IGN); + pqsignal(SIGALRM, PG_SIG_IGN); + pqsignal(SIGPIPE, PG_SIG_IGN); + pqsignal(SIGUSR1, procsignal_sigusr1_handler); + pqsignal(SIGUSR2, PG_SIG_IGN); + pqsignal(SIGCHLD, PG_SIG_DFL); + + /* Unblock signals (they were blocked when the postmaster forked us) */ + sigprocmask(SIG_SETMASK, &UnBlockSig, NULL); + + FastRecoveryMain(); + + proc_exit(0); +} diff --git a/src/backend/postmaster/launch_backend.c b/src/backend/postmaster/launch_backend.c index 8f3cfea880c..1a669d2da9e 100644 --- a/src/backend/postmaster/launch_backend.c +++ b/src/backend/postmaster/launch_backend.c @@ -54,6 +54,7 @@ #include "storage/shmem_internal.h" #include "tcop/backend_startup.h" #include "utils/memutils.h" +#include "postmaster/fast_recovery_worker.h" #ifdef EXEC_BACKEND #include "nodes/queryjumble.h" diff --git a/src/backend/postmaster/meson.build b/src/backend/postmaster/meson.build index 6cba23bbeef..ce7fb161377 100644 --- a/src/backend/postmaster/meson.build +++ b/src/backend/postmaster/meson.build @@ -7,6 +7,7 @@ backend_sources += files( 'bgwriter.c', 'checkpointer.c', 'datachecksum_state.c', + 'fast_recovery_worker.c', 'fork_process.c', 'interrupt.c', 'launch_backend.c', diff --git a/src/backend/postmaster/pmchild.c b/src/backend/postmaster/pmchild.c index 312cb2d8aba..42efa7bd534 100644 --- a/src/backend/postmaster/pmchild.c +++ b/src/backend/postmaster/pmchild.c @@ -128,6 +128,7 @@ InitPostmasterChildSlots(void) pmchild_pools[B_WAL_SUMMARIZER].size = 1; pmchild_pools[B_WAL_WRITER].size = 1; pmchild_pools[B_LOGGER].size = 1; + pmchild_pools[B_FAST_RECOVERY_WORKER].size = 1; /* The rest of the pmchild_pools are left at zero size */ diff --git a/src/backend/postmaster/postmaster.c b/src/backend/postmaster/postmaster.c index f53aa9c404d..08da3faff5e 100644 --- a/src/backend/postmaster/postmaster.c +++ b/src/backend/postmaster/postmaster.c @@ -123,6 +123,7 @@ #include "utils/pidfile.h" #include "utils/timestamp.h" #include "utils/varlena.h" +#include "access/lsn_indexer.h" #ifdef EXEC_BACKEND #include "common/file_utils.h" @@ -269,6 +270,9 @@ static PMChild *StartupPMChild = NULL, *SysLoggerPMChild = NULL, *SlotSyncWorkerPMChild = NULL; +PMChild *FastRecoveryWorkerPMChild = NULL; + + /* Startup process's status */ typedef enum { @@ -2362,6 +2366,8 @@ process_pm_child_exit(void) AbortStartTime = 0; UpdatePMState(PM_RUN); connsAllowed = true; + if (fast_crash_recovery && LSNIndexIsActive()) + FastRecoveryWorkerPMChild = StartChildProcess(B_FAST_RECOVERY_WORKER); /* * At the next iteration of the postmaster's main loop, we will @@ -2383,6 +2389,16 @@ process_pm_child_exit(void) continue; } + if (FastRecoveryWorkerPMChild && pid == FastRecoveryWorkerPMChild->pid) + { + ReleasePostmasterChildSlot(FastRecoveryWorkerPMChild); + FastRecoveryWorkerPMChild = NULL; + if (!EXIT_STATUS_0(exitstatus)) + HandleChildCrash(pid, exitstatus, + _("Fast Recovery Worker process")); + continue; + } + /* * Was it the bgwriter? Normal exit can be ignored; we'll start a new * one at the next iteration of the postmaster's main loop, if @@ -2976,6 +2992,14 @@ PostmasterStateMachine(void) B_STARTUP, B_WAL_RECEIVER); + /* + * Also wait for the fast recovery worker. It ignores SIGTERM: the + * shutdown checkpoint must not be written while pages are still + * unrecovered, so a smart or fast shutdown lets it finish first. + */ + targetMask = btmask_add(targetMask, + B_FAST_RECOVERY_WORKER); + /* * If we are doing crash recovery or an immediate shutdown then we * expect archiver, checkpointer, io workers and walsender to exit as @@ -3052,6 +3076,16 @@ PostmasterStateMachine(void) { SignalChildren(SIGTERM, targetMask); + /* + * The fast recovery worker ignores SIGTERM too, on purpose: a + * shutdown checkpoint must not be written while pages are + * still unrecovered, so we wait for it like a backend. + */ + if (FastRecoveryWorkerPMChild != NULL && + Shutdown < ImmediateShutdown) + ereport(LOG, + (errmsg("waiting for fast crash recovery to finish before shutting down"))); + UpdatePMState(PM_WAIT_BACKENDS); } } diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index f81c7732e68..b18ccbf1288 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -70,6 +70,7 @@ #include "utils/resowner.h" #include "utils/timestamp.h" #include "utils/wait_event.h" +#include "access/lsn_indexer.h" /* Note: these two macros only work on shared buffers, not local ones! */ @@ -1201,7 +1202,16 @@ ZeroAndLockBuffer(Buffer buffer, ReadBufferMode mode, bool already_valid) if (isLocalBuf) TerminateLocalBufferIO(bufHdr, false, BM_VALID, false); else + { TerminateBufferIO(bufHdr, false, BM_VALID, true, false); + + /* + * Fast crash recovery: WAL still pending for this page no longer + * applies, now that the caller initializes it from scratch. + */ + if (unlikely(LSNIndexIsActive())) + LSNIndexForgetPage(&bufHdr->tag); + } } else if (!isLocalBuf) { @@ -1369,6 +1379,100 @@ ReadBuffer_common(Relation rel, SMgrRelation smgr, char smgr_persistence, return buffer; } +/* + * Fast crash recovery: read a block that still has pending WAL records, and + * replay them before anyone else can see the page. + * + * The whole job is done under the buffer's I/O-in-progress flag, the same + * protocol a read uses, so a backend that wants the page meanwhile waits in + * WaitIO() until it's valid. The page is thus private to us while redo runs: + * redo needs no lock to keep readers out, no reader can be holding a pointer + * into the page (which is what cleanup locks guarantee), and every way of + * reading pages (ReadBuffer, read streams, hits) only ever sees recovered + * ones. With AIO a read completes inside a critical section, possibly in + * another process, where redo can't run; so pages with pending records are + * read synchronously here instead. + * + * Returns false if the block has nothing pending, in which case the caller + * reads it as usual; true once the buffer is valid. + * + * If anything in here fails, the error paths (transaction abort, or process + * exit) call LSNIndexErrorCleanup(), which cleans the buffer through + * AbortPendingBufferReplay() below and resets the replay state; the resource + * owner then ends the I/O as for any failed read. + */ +static bool +ReadAndReplayPendingBuffer(Buffer buffer, SMgrRelation smgr, + ForkNumber forknum, BlockNumber blocknum, + IOObject io_object, IOContext io_context) +{ + BufferDesc *bufHdr = GetBufferDescriptor(buffer - 1); + + if (!LSNIndexBeginPageReplay(&bufHdr->tag)) + return false; + + /* + * A read stream may call us in AIO batch mode, with IOs staged but not + * submitted. We're about to wait for other backends and to read other + * pages, so submit them first; see pgaio_enter_batchmode(). + */ + pgaio_submit_staged(); + + if (StartSharedBufferIO(bufHdr, true, true, NULL) == BUFFER_IO_READY_FOR_IO) + { + Block bufBlock = BufHdrGetBlock(bufHdr); + instr_time io_start; + + io_start = pgstat_prepare_io_time(track_io_timing); + smgrread(smgr, forknum, blocknum, bufBlock); + pgstat_count_io_op_time(io_object, io_context, IOOP_READ, + io_start, 1, BLCKSZ); + pgBufferUsage.shared_blks_read++; + + if (!PageIsVerified((Page) bufBlock, blocknum, PIV_LOG_WARNING, NULL)) + ereport(ERROR, + (errcode(ERRCODE_DATA_CORRUPTED), + errmsg("invalid page in block %u of relation \"%s\"", + blocknum, + relpathperm(smgr->smgr_rlocator.locator, + forknum).str))); + + LSNIndexReplayIntoBuffer(buffer); + + TerminateBufferIO(bufHdr, false, BM_VALID, true, false); + + /* Valid means current: never replay it again. */ + LSNIndexForgetPage(&bufHdr->tag); + } + /* else someone else replayed it while we waited */ + + LSNIndexEndPageReplay(); + + return true; +} + +/* + * Fast crash recovery: the on-demand replay into 'buffer' failed. + * + * Called by LSNIndexErrorCleanup() before the resource owner that started + * the I/O gets to AbortBufferIO(), which ends it as for any failed read. + * Redo dirtied the buffer before it was valid, which nothing else ever does, + * and AbortBufferIO() rightly insists that an invalid buffer is clean; so + * clean it, and leave the rest to the usual path. The page was never valid + * and its entry is still in the index: the next reader starts over. + */ +void +AbortPendingBufferReplay(Buffer buffer) +{ + BufferDesc *bufHdr = GetBufferDescriptor(buffer - 1); + uint64 buf_state; + + buf_state = LockBufHdr(bufHdr); + Assert(buf_state & BM_IO_IN_PROGRESS); + Assert(!(buf_state & BM_VALID)); + UnlockBufHdrExt(bufHdr, buf_state, 0, BM_DIRTY | BM_CHECKPOINT_NEEDED, 0); +} + static pg_always_inline bool StartReadBuffersImpl(ReadBuffersOperation *operation, Buffer *buffers, @@ -1456,6 +1560,17 @@ StartReadBuffersImpl(ReadBuffersOperation *operation, &found); } + /* + * Fast crash recovery: a page with pending WAL records must not + * become valid before they're applied. Read and replay it right + * here; then it's a hit like any other. + */ + if (!found && unlikely(LSNIndexIsActive()) && + operation->persistence == RELPERSISTENCE_PERMANENT) + found = ReadAndReplayPendingBuffer(buffers[i], operation->smgr, + operation->forknum, blockNum + i, + io_object, io_context); + if (found) { /* diff --git a/src/backend/storage/buffer/localbuf.c b/src/backend/storage/buffer/localbuf.c index 4870c8e13d0..3ede8191199 100644 --- a/src/backend/storage/buffer/localbuf.c +++ b/src/backend/storage/buffer/localbuf.c @@ -15,6 +15,7 @@ */ #include "postgres.h" +#include "access/lsn_indexer.h" #include "access/parallel.h" #include "executor/instrument.h" #include "pgstat.h" @@ -762,8 +763,13 @@ InitLocalBuffers(void) * that we don't wish to prevent a parallel worker from accessing catalog * metadata about a temp table, so checks at higher levels would be * inappropriate. + * + * Fast crash recovery is the exception: replaying a page on demand, which + * a parallel worker does when it reads one, uses local buffers as private + * scratch space for the record's other pages (see + * LSNIndexScratchBuffer()), never for a temporary table. */ - if (IsParallelWorker()) + if (IsParallelWorker() && !inReplayPageWals) ereport(ERROR, (errcode(ERRCODE_INVALID_TRANSACTION_STATE), errmsg("cannot access temporary tables during a parallel operation"))); diff --git a/src/backend/storage/freespace/freespace.c b/src/backend/storage/freespace/freespace.c index 40d67a96178..134db5e3808 100644 --- a/src/backend/storage/freespace/freespace.c +++ b/src/backend/storage/freespace/freespace.c @@ -24,6 +24,7 @@ #include "postgres.h" #include "access/htup_details.h" +#include "access/lsn_indexer.h" #include "access/xloginsert.h" #include "access/xlogutils.h" #include "miscadmin.h" @@ -218,6 +219,17 @@ XLogRecordPageWithFreeSpace(RelFileLocator rlocator, BlockNumber heapBlk, Buffer buf; Page page; + /* + * During on-demand replay of a heap page, leave the free space map alone. + * Reading the map page here could find it with pending records of its own + * (full-page images for hints), and replaying that page inside this one's + * redo is the kind of nesting on-demand replay doesn't do. The map is + * advisory: an insert that finds less space than recorded corrects the + * entry, and VACUUM rebuilds it. + */ + if (inReplayPageWals) + return; + /* Get the location of the FSM byte representing the heap block */ addr = fsm_get_location(heapBlk, &slot); blkno = fsm_logical_to_physical(addr); diff --git a/src/backend/storage/ipc/ipci.c b/src/backend/storage/ipc/ipci.c index e149a738c8d..132c362969a 100644 --- a/src/backend/storage/ipc/ipci.c +++ b/src/backend/storage/ipc/ipci.c @@ -24,6 +24,7 @@ #include "storage/shmem_internal.h" #include "storage/subsystems.h" #include "utils/guc.h" +#include "access/lsn_indexer.h" /* GUCs */ int shared_memory_type = DEFAULT_SHARED_MEMORY_TYPE; diff --git a/src/backend/utils/activity/pgstat_backend.c b/src/backend/utils/activity/pgstat_backend.c index b736b2ccc6f..99730bf2f76 100644 --- a/src/backend/utils/activity/pgstat_backend.c +++ b/src/backend/utils/activity/pgstat_backend.c @@ -455,6 +455,7 @@ pgstat_tracks_backend_bktype(BackendType bktype) case B_STARTUP: case B_DATACHECKSUMSWORKER_LAUNCHER: case B_DATACHECKSUMSWORKER_WORKER: + case B_FAST_RECOVERY_WORKER: return false; case B_AUTOVAC_WORKER: diff --git a/src/backend/utils/activity/pgstat_io.c b/src/backend/utils/activity/pgstat_io.c index 8ec1aad5078..3cfec87b39b 100644 --- a/src/backend/utils/activity/pgstat_io.c +++ b/src/backend/utils/activity/pgstat_io.c @@ -16,6 +16,7 @@ #include "postgres.h" +#include "access/lsn_indexer.h" #include "executor/instrument.h" #include "storage/bufmgr.h" #include "utils/pgstat_internal.h" @@ -370,6 +371,7 @@ pgstat_tracks_io_bktype(BackendType bktype) case B_WAL_SENDER: case B_WAL_SUMMARIZER: case B_WAL_WRITER: + case B_FAST_RECOVERY_WORKER: return true; } @@ -487,6 +489,23 @@ pgstat_tracks_io_object(BackendType bktype, IOObject io_object, return false; } + /* + * The fast recovery worker reads relation pages into shared buffers, + * evicting dirty ones as needed, and reads WAL to replay them on demand. + * Replaying a record can hand its other pages scratch local buffers. It + * never uses a buffer access strategy and never initializes WAL segments. + */ + if (bktype == B_FAST_RECOVERY_WORKER) + { + if (io_context == IOCONTEXT_NORMAL && + (io_object == IOOBJECT_RELATION || + io_object == IOOBJECT_TEMP_RELATION || + io_object == IOOBJECT_WAL)) + return true; + + return false; + } + return true; } @@ -525,9 +544,12 @@ pgstat_tracks_io_op(BackendType bktype, IOObject io_object, return false; /* - * Some BackendTypes do not perform reads with IOOBJECT_WAL. + * Some BackendTypes do not perform reads with IOOBJECT_WAL - except when + * fast crash recovery is on, in which case any backend that hits a cold + * page may have to read WAL to replay it on demand. */ if (io_object == IOOBJECT_WAL && io_op == IOOP_READ && + !fast_crash_recovery && (bktype == B_WAL_RECEIVER || bktype == B_BG_WRITER || bktype == B_AUTOVAC_LAUNCHER || bktype == B_AUTOVAC_WORKER || bktype == B_DATACHECKSUMSWORKER_LAUNCHER || diff --git a/src/backend/utils/activity/wait_event_names.txt b/src/backend/utils/activity/wait_event_names.txt index 3d366fd1114..2abef752587 100644 --- a/src/backend/utils/activity/wait_event_names.txt +++ b/src/backend/utils/activity/wait_event_names.txt @@ -119,6 +119,7 @@ CHECKPOINT_START "Waiting for a checkpoint to start." CHECKSUM_ENABLE_STARTCONDITION "Waiting for data checksums enabling to start." CHECKSUM_ENABLE_TEMPTABLE_WAIT "Waiting for temporary tables to be dropped for data checksums to be enabled." EXECUTE_GATHER "Waiting for activity from a child process while executing a Gather plan node." +FAST_RECOVERY_DRAIN "Waiting for on-demand WAL replays still in progress to finish before fast crash recovery completes." HASH_BATCH_ALLOCATE "Waiting for an elected Parallel Hash participant to allocate a hash table." HASH_BATCH_ELECT "Waiting to elect a Parallel Hash participant to allocate a hash table." HASH_BATCH_LOAD "Waiting for other Parallel Hash participants to finish loading a hash table." @@ -372,6 +373,7 @@ LogicalDecodingControl "Waiting to read or update logical decoding status inform DataChecksumsWorker "Waiting for data checksums worker." AioWorkerControl "Waiting to update AIO worker information." DataChecksumTransition "Waiting for a data checksum state transition to be written to WAL." +LSNIndex "Waiting to create or attach the fast-recovery LSN index." # # END OF PREDEFINED LWLOCKS (DO NOT CHANGE THIS LINE) @@ -419,6 +421,8 @@ XactSLRU "Waiting to access the transaction status SLRU cache." ParallelVacuumDSA "Waiting for parallel vacuum dynamic shared memory allocation." AioUringCompletion "Waiting for another process to complete IO via io_uring." ShmemIndex "Waiting to find or allocate space in shared memory." +LSNIndexDSA "Waiting for fast recovery LSN index dynamic shared memory allocation." +LSNIndexHash "Waiting to access the fast recovery LSN index hash table." # No "ABI_compatibility" region here as WaitEventLWLock has its own C code. diff --git a/src/backend/utils/misc/guc_parameters.dat b/src/backend/utils/misc/guc_parameters.dat index 380679a76c0..acbbde0a181 100644 --- a/src/backend/utils/misc/guc_parameters.dat +++ b/src/backend/utils/misc/guc_parameters.dat @@ -1083,6 +1083,13 @@ max => '3', }, +{ name => 'fast_crash_recovery', type => 'bool', context => 'PGC_POSTMASTER', group => 'WAL_RECOVERY', + short_desc => 'Defers WAL replay in crash recovery until pages are first read.', + long_desc => 'The server accepts connections once the WAL has been scanned; each page is recovered when it is first read, and a background worker recovers the rest.', + variable => 'fast_crash_recovery', + boot_val => 'false', +}, + { name => 'file_copy_method', type => 'enum', context => 'PGC_USERSET', group => 'RESOURCES_DISK', short_desc => 'Selects the file copy method.', variable => 'file_copy_method', diff --git a/src/backend/utils/misc/guc_tables.c b/src/backend/utils/misc/guc_tables.c index 342aaeef59a..20e524a0e53 100644 --- a/src/backend/utils/misc/guc_tables.c +++ b/src/backend/utils/misc/guc_tables.c @@ -105,6 +105,7 @@ #include "utils/ps_status.h" #include "utils/rls.h" #include "utils/xml.h" +#include "access/lsn_indexer.h" #ifdef TRACE_SYNCSCAN #include "access/syncscan.h" diff --git a/src/backend/utils/misc/postgresql.conf.sample b/src/backend/utils/misc/postgresql.conf.sample index 01f98de8a9f..4820c3f8244 100644 --- a/src/backend/utils/misc/postgresql.conf.sample +++ b/src/backend/utils/misc/postgresql.conf.sample @@ -344,6 +344,11 @@ #summarize_wal = off # run WAL summarizer process? #wal_summary_keep_time = '10d' # when to remove old summary files, 0 = never +# - Fast Recovery - + +#fast_crash_recovery = off # replay WAL on demand after a crash, + # instead of before accepting connections + # (change requires restart) #------------------------------------------------------------------------------ # REPLICATION diff --git a/src/include/access/lsn_indexer.h b/src/include/access/lsn_indexer.h new file mode 100644 index 00000000000..8051f7046f0 --- /dev/null +++ b/src/include/access/lsn_indexer.h @@ -0,0 +1,142 @@ +/*------------------------------------------------------------------------- + * + * lsn_indexer.h + * Index of pending WAL records by page, for on-demand replay after a crash + * + * The index maps a BufferTag to the list of LSNs of the WAL records that + * touch that page and have not been replayed yet. It lives in a dshash + * table in a DSA area, so it needs no size estimate up front. + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * + * src/include/access/lsn_indexer.h + * + *------------------------------------------------------------------------- + */ + +#ifndef LSN_INDEXER_H +#define LSN_INDEXER_H + +#include "access/xlogreader.h" +#include "lib/dshash.h" +#include "port/atomics.h" +#include "storage/block.h" +#include "storage/buf_internals.h" +#include "utils/dsa.h" + +/* + * LSNNode - one WAL record LSN for a page, stored in DSA memory. + * Linked via dsa_pointer instead of raw pointers. + */ +typedef struct LSNNode +{ + XLogRecPtr lsn; + dsa_pointer next; /* dsa_pointer to next LSNNode, or + * InvalidDsaPointer */ +} LSNNode; + +/* + * PageLSNEntry - dshash entry: BufferTag -> list of LSNs. + * The key (BufferTag) must be at the start for dshash. + */ +typedef struct PageLSNEntry +{ + BufferTag tag; + dsa_pointer lsn_head; /* dsa_pointer to first LSNNode */ + dsa_pointer lsn_tail; /* dsa_pointer to last LSNNode */ + XLogRecPtr max_lsn; +} PageLSNEntry; + +/* + * LSNIndexControl - small fixed-size struct in traditional shared memory. + * Holds handles so any process can attach to the DSA and dshash. + * + * is_active is set when the startup process creates the index and cleared + * by LSNIndexFinish(), both under LSNIndexLock. nreplaying counts pages + * being replayed right now, between LSNIndexBeginPageReplay() and + * LSNIndexEndPageReplay(); it is only incremented while holding LSNIndexLock + * in shared mode and seeing is_active set. + */ +typedef struct LSNIndexControl +{ + dsa_handle dsa_handle; + dshash_table_handle hash_handle; + bool is_active; + pg_atomic_uint32 nreplaying; +} LSNIndexControl; + +/* GUC */ +extern bool fast_crash_recovery; + +/* Shared control structure (in traditional shmem) */ +extern LSNIndexControl *LSNIndexCtl; + +/* + * While a page is being replayed on demand: inReplayPageWals is true, + * targetTag is the page, and targetBuffer the buffer it's being replayed + * into. XLogReadBufferExtended() uses them to hand redo that buffer, + * XLogReadBufferForRedoExtended() to skip the record's other pages, and + * XLogRecordPageWithFreeSpace() to leave the free space map alone. Replays + * never nest, so a single set of these is enough per process. + */ +extern bool inReplayPageWals; +extern BufferTag targetTag; +extern Buffer targetBuffer; + +/* Phase 1: postmaster-safe - allocates control struct in main shmem */ +extern Size LSNIndexShmemSize(void); +extern void LSNIndexShmemInit(void *arg); + +/* + * Phase 2: backend-context - creates the DSA+dshash on first call, + * attaches on subsequent calls. Called by the startup process. + */ +extern void LSNIndexInit(void); + +/* Attach to an existing index - used by backends and the worker */ +extern void LSNIndexAttach(void); +extern void LSNIndexDetach(void); + +/* Check if the LSN index is active */ +extern bool LSNIndexIsActive(void); + +/* May pre-crash WAL still be replayed onto some page? Blocks checkpoints. */ +extern bool FastRecoveryInProgress(void); + +/* Add an entry during WAL scan (startup process) */ +extern void LSNIndexAddEntry(XLogReaderState *state); + +/* May the startup process index this record? Evicts resident pages for it. */ +extern bool LSNIndexPrepareToDefer(XLogReaderState *record); + +/* Report what the scan did, once it is over */ +extern void LSNIndexLogScanSummary(void); + +/* Forget pages removed by a drop or truncation replayed during WAL scan */ +extern void LSNIndexForgetRelation(RelFileLocator rlocator, ForkNumber forknum, + BlockNumber minblkno); +extern void LSNIndexForgetDatabase(Oid dbid); + +/* + * On-demand replay of one page, driven by the buffer manager while it holds + * the page's I/O-in-progress flag (see ReadAndReplayPendingBuffer()). + */ +extern bool LSNIndexBeginPageReplay(const BufferTag *tag); +extern void LSNIndexReplayIntoBuffer(Buffer buffer); +extern void LSNIndexEndPageReplay(void); +extern void LSNIndexErrorCleanup(void); + +/* Drop a page's pending records: replayed, or about to be zeroed */ +extern void LSNIndexForgetPage(const BufferTag *tag); + +/* Scratch buffer for a page a record touches besides the one being replayed */ +extern Buffer LSNIndexScratchBuffer(RelFileLocator rlocator, ForkNumber forknum, + BlockNumber blkno); + +/* Get the dshash table pointer (for sequential scans by the worker) */ +extern dshash_table *LSNIndexGetHash(void); + +/* End fast recovery once every page has been replayed (worker only) */ +extern void LSNIndexFinish(void); + +#endif /* LSN_INDEXER_H */ diff --git a/src/include/access/xlog.h b/src/include/access/xlog.h index 2bf5714b51e..62ca52c93a0 100644 --- a/src/include/access/xlog.h +++ b/src/include/access/xlog.h @@ -345,6 +345,7 @@ extern void do_pg_backup_stop(BackupState *state, bool waitforarchive); extern void do_pg_abort_backup(int code, Datum arg); extern void register_persistent_abort_backup_handler(void); extern SessionBackupState get_backup_status(void); +extern int get_control_state(void); /* File path names (all relative to $PGDATA) */ #define RECOVERY_SIGNAL_FILE "recovery.signal" diff --git a/src/include/miscadmin.h b/src/include/miscadmin.h index 8d6aacc4d5a..078587edacd 100644 --- a/src/include/miscadmin.h +++ b/src/include/miscadmin.h @@ -369,6 +369,7 @@ typedef enum BackendType B_WAL_RECEIVER, B_WAL_SUMMARIZER, B_WAL_WRITER, + B_FAST_RECOVERY_WORKER, /* * Data checksums processes are dynamic background workers, but they use diff --git a/src/include/postmaster/fast_recovery_worker.h b/src/include/postmaster/fast_recovery_worker.h new file mode 100644 index 00000000000..19725306a8d --- /dev/null +++ b/src/include/postmaster/fast_recovery_worker.h @@ -0,0 +1,17 @@ +/*------------------------------------------------------------------------- + * + * fast_recovery_worker.h + * Exports from postmaster/fast_recovery_worker.c + * + * Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group + * + * src/include/postmaster/fast_recovery_worker.h + * + *------------------------------------------------------------------------- + */ +#ifndef FAST_RECOVERY_WORKER_H +#define FAST_RECOVERY_WORKER_H + +extern void FastRecoveryWorkerMain(const void *startup_data, size_t startup_data_len); + +#endif /* FAST_RECOVERY_WORKER_H */ diff --git a/src/include/postmaster/postmaster.h b/src/include/postmaster/postmaster.h index 716b4c912b3..44dcf451b3d 100644 --- a/src/include/postmaster/postmaster.h +++ b/src/include/postmaster/postmaster.h @@ -47,6 +47,8 @@ typedef struct dlist_node elem; /* list link in ActiveChildList */ } PMChild; +extern PMChild *FastRecoveryWorkerPMChild; + #ifdef EXEC_BACKEND extern PGDLLIMPORT int num_pmchild_slots; #endif diff --git a/src/include/postmaster/proctypelist.h b/src/include/postmaster/proctypelist.h index 152fde394f4..74204b26768 100644 --- a/src/include/postmaster/proctypelist.h +++ b/src/include/postmaster/proctypelist.h @@ -41,6 +41,7 @@ PG_PROCTYPE(B_CHECKPOINTER, "checkpointer", gettext_noop("checkpointer"), Checkp PG_PROCTYPE(B_DATACHECKSUMSWORKER_LAUNCHER, "checksums", gettext_noop("datachecksums launcher"), NULL, false) PG_PROCTYPE(B_DATACHECKSUMSWORKER_WORKER, "checksums", gettext_noop("datachecksums worker"), NULL, false) PG_PROCTYPE(B_DEAD_END_BACKEND, "backend", gettext_noop("dead-end client backend"), BackendMain, true) +PG_PROCTYPE(B_FAST_RECOVERY_WORKER, "fastrecoveryworker", gettext_noop("fast recovery worker"), FastRecoveryWorkerMain, true) PG_PROCTYPE(B_INVALID, "postmaster", gettext_noop("unrecognized"), NULL, false) PG_PROCTYPE(B_IO_WORKER, "ioworker", gettext_noop("io worker"), IoWorkerMain, true) PG_PROCTYPE(B_LOGGER, "syslogger", gettext_noop("syslogger"), SysLoggerMain, false) diff --git a/src/include/storage/bufmgr.h b/src/include/storage/bufmgr.h index 6837b35fc6d..33d0e4295c0 100644 --- a/src/include/storage/bufmgr.h +++ b/src/include/storage/bufmgr.h @@ -358,6 +358,9 @@ extern bool EvictUnpinnedBuffer(Buffer buf, bool *buffer_flushed); extern void EvictAllUnpinnedBuffers(int32 *buffers_evicted, int32 *buffers_flushed, int32 *buffers_skipped); + +/* fast crash recovery: clean a buffer whose on-demand replay failed */ +extern void AbortPendingBufferReplay(Buffer buffer); extern void EvictRelUnpinnedBuffers(Relation rel, int32 *buffers_evicted, int32 *buffers_flushed, diff --git a/src/include/storage/lwlocklist.h b/src/include/storage/lwlocklist.h index 8d858be9927..49dda69139c 100644 --- a/src/include/storage/lwlocklist.h +++ b/src/include/storage/lwlocklist.h @@ -90,6 +90,7 @@ PG_LWLOCK(55, LogicalDecodingControl) PG_LWLOCK(56, DataChecksumsWorker) PG_LWLOCK(57, AioWorkerControl) PG_LWLOCK(58, DataChecksumTransition) +PG_LWLOCK(59, LSNIndex) /* * There also exist several built-in LWLock tranches. As with the predefined @@ -141,3 +142,5 @@ PG_LWLOCKTRANCHE(XACT_SLRU, XactSLRU) PG_LWLOCKTRANCHE(PARALLEL_VACUUM_DSA, ParallelVacuumDSA) PG_LWLOCKTRANCHE(AIO_URING_COMPLETION, AioUringCompletion) PG_LWLOCKTRANCHE(SHMEM_INDEX, ShmemIndex) +PG_LWLOCKTRANCHE(LSN_INDEX_DSA, LSNIndexDSA) +PG_LWLOCKTRANCHE(LSN_INDEX_HASH, LSNIndexHash) diff --git a/src/include/storage/subsystemlist.h b/src/include/storage/subsystemlist.h index 9ad619080be..13a8db9fce6 100644 --- a/src/include/storage/subsystemlist.h +++ b/src/include/storage/subsystemlist.h @@ -88,3 +88,4 @@ PG_SHMEM_SUBSYSTEM(DataChecksumsShmemCallbacks) /* AIO subsystem. This delegates to the method-specific callbacks */ PG_SHMEM_SUBSYSTEM(AioShmemCallbacks) +PG_SHMEM_SUBSYSTEM(LsnIndexerShmemCallbacks) diff --git a/src/test/regress/expected/stats.out b/src/test/regress/expected/stats.out index 8b15471248b..97e3f809b36 100644 --- a/src/test/regress/expected/stats.out +++ b/src/test/regress/expected/stats.out @@ -60,6 +60,9 @@ datachecksums worker|relation|normal datachecksums worker|relation|vacuum datachecksums worker|wal|init datachecksums worker|wal|normal +fast recovery worker|relation|normal +fast recovery worker|temp relation|normal +fast recovery worker|wal|normal io worker|relation|bulkread io worker|relation|bulkwrite io worker|relation|init @@ -104,7 +107,7 @@ walsummarizer|wal|init walsummarizer|wal|normal walwriter|wal|init walwriter|wal|normal -(88 rows) +(91 rows) \a -- List of registered statistics kinds. SELECT id, name, fixed_amount, diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index 3bed561f7f2..5ac73adc737 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -1701,6 +1701,8 @@ LogicalSlotInfoArr LogicalTape LogicalTapeSet LookupSet +LSNIndexControl +LSNNode LsnReadQueue LsnReadQueueNextFun LsnReadQueueNextStatus @@ -1905,6 +1907,7 @@ OutputPluginOutputType OverridingKind PACE_HEADER PACL +PageLSNEntry PATH PCtxtHandle PERL_CONTEXT -- 2.43.0