From 329939415d2f4e557aed20d0c70de357a76de732 Mon Sep 17 00:00:00 2001 From: Ayush Tiwari Date: Wed, 7 Oct 2026 12:48:00 +0000 Subject: [PATCH v1] Reset unlogged relations before syncing the data directory Crash recovery currently syncs disposable unlogged relation forks before removing them, causing unnecessary writeback of dirty cached contents. After InitWalRecovery() has rebuilt any tablespace links, remove these forks before SyncDataDirectory() on crash starts. Keep init forks and all surviving files synced before WAL replay, and leave the end-of-recovery checkpoint unchanged. For recovery following a clean shutdown, retain the later cleanup after pg_control records recovery. Otherwise, a failed startup could delete unlogged files while pg_control still permits a restart without recovery. --- src/backend/access/transam/xlog.c | 41 +++++++++++++++++++------------ 1 file changed, 25 insertions(+), 16 deletions(-) diff --git a/src/backend/access/transam/xlog.c b/src/backend/access/transam/xlog.c index afbe068cd0a..85955c1eb5d 100644 --- a/src/backend/access/transam/xlog.c +++ b/src/backend/access/transam/xlog.c @@ -6179,25 +6179,16 @@ StartupXLOG(void) RegisterTimeout(STARTUP_PROGRESS_TIMEOUT, startup_progress_timeout_handler); - /*---------- - * If we previously crashed, perform a couple of actions: - * - * - The pg_wal directory may still include some temporary WAL segments - * used when creating a new segment, so perform some clean up to not - * bloat this path. This is done first as there is no point to sync - * this temporary data. - * - * - There might be data which we had written, intending to fsync it, but - * which we had not actually fsync'd yet. Therefore, a power failure in - * the near future might cause earlier unflushed writes to be lost, even - * though more recent data written to disk from here on would be - * persisted. To avoid that, fsync the entire data directory. + /* + * If we previously crashed, the pg_wal directory may still include some + * temporary WAL segments used when creating a new segment, so perform + * some clean up to not bloat this path. This is done first as there is + * no point to sync this temporary data. */ if (ControlFile->state != DB_SHUTDOWNED && ControlFile->state != DB_SHUTDOWNED_IN_RECOVERY) { RemoveTempXlogFiles(); - SyncDataDirectory(); didCrash = true; } else @@ -6215,6 +6206,22 @@ StartupXLOG(void) &haveBackupLabel, &haveTblspcMap); checkPoint = ControlFile->checkPointCopy; + /* + * If we previously crashed, there might be data which we had written, + * intending to fsync it, but which we had not actually fsync'd yet. + * Therefore, a power failure in the near future might cause earlier + * unflushed writes to be lost, even though more recent data written to + * disk from here on would be persisted. To avoid that, fsync the entire + * data directory, after removing unlogged relations' disposable forks. + * Cleanup needs the tablespace links restored by InitWalRecovery(); the + * on-disk pg_control state already requires recovery on this path. + */ + if (didCrash) + { + ResetUnloggedRelations(UNLOGGED_RELATION_CLEANUP); + SyncDataDirectory(); + } + /* initialize shared memory variables from the checkpoint record */ TransamVariables->nextXid = checkPoint.nextXid; TransamVariables->nextOid = checkPoint.nextOid; @@ -6460,9 +6467,11 @@ StartupXLOG(void) * We're in recovery, so unlogged relations may be trashed and must be * reset. This should be done BEFORE allowing Hot Standby * connections, so that read-only backends don't try to read whatever - * garbage is left over from before. + * garbage is left over from before. After a crash, cleanup was + * already done before SyncDataDirectory(). */ - ResetUnloggedRelations(UNLOGGED_RELATION_CLEANUP); + if (!didCrash) + ResetUnloggedRelations(UNLOGGED_RELATION_CLEANUP); /* * Likewise, delete any saved transaction snapshot files that got left base-commit: 1f2245371e31f0def347546f5ad250552a4eb8a4 -- 2.43.0