From ed05777dfa6f0a7ca766d613b877c3a01f4920b2 Mon Sep 17 00:00:00 2001
From: Gustavo William <gustavo.oliveira@enterprisedb.com>
Date: Tue, 15 Sep 2026 13:15:51 -0400
Subject: [PATCH v06102026 05/10] basebackup: issue posix_fadvise() for more
 efficient cold-cache handling

For full backups, inform the OS that we are going to need the whole
file being backed up, so it can perform more efficient read-ahead.
This in turn allows our synchronous basebackup_read_file()->pread64()
calls to see lower latencies, since the data is served from the page
cache more often, increasing the overall throughput of the transfer.

For incremental backups, issue posix_fadvise(POSIX_FADV_WILLNEED)
hints for upcoming blocks before the read loop reaches them, giving
the kernel a chance to fetch them into the page cache asynchronously
while we're still processing the current block. How far ahead we
prefetch is bounded by the maintenance_io_concurrency GUC. Contiguous
blocks are merged into a single read-ahead request to cut down on the
number of syscalls and improve I/O efficiency.

Co-authored-by: Gustavo William <gustavo.oliveira@enterprisedb.com>
Co-authored-by: Jakub Wartak <jakub.wartak@enterprisedb.com>
Reviewed-by:
Discussion: https://postgr.es/m/CAKZiRmwwW-hDc3B6ERJB%2BpaX7RNSBcQLheq1KdsTf42cGuRvuA%40mail.gmail.com
---
 src/backend/backup/basebackup.c | 94 ++++++++++++++++++++++++++++++++-
 1 file changed, 93 insertions(+), 1 deletion(-)

diff --git a/src/backend/backup/basebackup.c b/src/backend/backup/basebackup.c
index 6478218cc91..e1039569c82 100644
--- a/src/backend/backup/basebackup.c
+++ b/src/backend/backup/basebackup.c
@@ -12,6 +12,7 @@
  */
 #include "postgres.h"
 
+#include <fcntl.h>
 #include <sys/stat.h>
 #include <unistd.h>
 #include <time.h>
@@ -38,6 +39,7 @@
 #include "replication/slot.h"
 #include "replication/walsender.h"
 #include "replication/walsender_private.h"
+#include "storage/bufmgr.h"
 #include "storage/bufpage.h"
 #include "storage/checksum.h"
 #include "storage/dsm_impl.h"
@@ -106,6 +108,13 @@ static off_t read_file_data_into_buffer(bbsink *sink,
 										BlockNumber blkno,
 										bool verify_checksum,
 										int *checksum_failures);
+#if defined(USE_POSIX_FADVISE) && defined(POSIX_FADV_WILLNEED)
+static void prefetch_next_incremental_run(int fd,
+										  BlockNumber *incremental_blocks,
+										  unsigned num_incremental_blocks,
+										  unsigned max_run_len,
+										  unsigned *prefetch_idx);
+#endif
 static void push_to_sink(bbsink *sink, pg_checksum_context *checksum_ctx,
 						 size_t *bytes_done, void *data, size_t length);
 static bool backup_checksums_verifiable(XLogRecPtr start_lsn);
@@ -1591,7 +1600,11 @@ sendFile(bbsink *sink, const char *readfilename, const char *tarfilename,
 	pgoff_t		bytes_done = 0;
 	bool		verify_checksum = false;
 	pg_checksum_context checksum_ctx;
-	int			ibindex = 0;
+	unsigned	ibindex = 0;
+#if defined(USE_POSIX_FADVISE) && defined(POSIX_FADV_WILLNEED)
+	unsigned	prefetch_idx = 0;
+	unsigned	prefetch_window_size = maintenance_io_concurrency;
+#endif
 
 	if (pg_checksum_init(&checksum_ctx, manifest->checksum_type) < 0)
 		elog(ERROR, "could not initialize checksum of file \"%s\"",
@@ -1607,6 +1620,36 @@ sendFile(bbsink *sink, const char *readfilename, const char *tarfilename,
 				 errmsg("could not open file \"%s\": %m", readfilename)));
 	}
 
+	/*
+	 * On full backups, let the OS know that we are going to read the whole file.
+	 * It's just an hint, but it helps avoid longer synchronous read stalls.
+	 */
+#if defined(USE_POSIX_FADVISE) && defined(POSIX_FADV_SEQUENTIAL)
+	if (incremental_blocks == NULL)
+		(void) posix_fadvise(fd, 0, 0, POSIX_FADV_SEQUENTIAL);
+#endif
+
+	/*
+	 * On incremental backups, let the OS know the exact blocks we'll need.
+	 * For that, maintain a bounded window of readahead hints, sized by
+	 * maintenance_io_concurrency: we keep at least N runs fadvised ahead
+	 * of what we're currently reading (N = maintenance_io_concurrency),
+	 * which increases I/O concurrency.
+	 *
+	 * Initially, advance the window by triggering prefetch_window_size runs.
+	*/
+#if defined(USE_POSIX_FADVISE) && defined(POSIX_FADV_WILLNEED)
+	if (incremental_blocks != NULL && prefetch_window_size > 1)
+	{
+		int			initial_prefetch = prefetch_window_size;
+
+		while (initial_prefetch-- > 0 && prefetch_idx < num_incremental_blocks)
+			prefetch_next_incremental_run(fd, incremental_blocks,
+										  num_incremental_blocks,
+										  prefetch_window_size, &prefetch_idx);
+	}
+#endif
+
 	_tarWriteHeader(sink, tarfilename, NULL, statbuf, false);
 
 	/*
@@ -1730,6 +1773,7 @@ sendFile(bbsink *sink, const char *readfilename, const char *tarfilename,
 			 * supposed to include.
 			 */
 			relative_blkno = incremental_blocks[ibindex++];
+
 			cnt = read_file_data_into_buffer(sink, readfilename, fd,
 											 relative_blkno * BLCKSZ,
 											 BLCKSZ,
@@ -1737,6 +1781,15 @@ sendFile(bbsink *sink, const char *readfilename, const char *tarfilename,
 											 verify_checksum,
 											 &checksum_failures);
 
+#if defined(USE_POSIX_FADVISE) && defined(POSIX_FADV_WILLNEED)
+			/* Advance the prefetch window by prefetching another run */
+			if (prefetch_window_size > 1 && prefetch_idx < num_incremental_blocks)
+				prefetch_next_incremental_run(fd, incremental_blocks,
+											  num_incremental_blocks,
+											  prefetch_window_size,
+											  &prefetch_idx);
+#endif
+
 			/*
 			 * If we get a partial read, that must mean that the relation is
 			 * being truncated. Ultimately, it should be truncated to a
@@ -1840,6 +1893,45 @@ sendFile(bbsink *sink, const char *readfilename, const char *tarfilename,
 	return true;
 }
 
+#if defined(USE_POSIX_FADVISE) && defined(POSIX_FADV_WILLNEED)
+
+/*
+ * Prefetch the next run of blocks of a file that will be needed soon.
+ *
+ * Advance *prefetch_idx past the next run of block numbers in incremental_blocks
+ * and issue a single readahead hint covering that whole run. This is advisory only:
+ * posix_fadvise() failures are ignored, since the worst that happens is that we
+ * don't get the intended prefetching benefit.
+ *
+ * Note that a "run" does not mean a "block": contiguous blocks are merged into
+ * the same run (up to max_run_len blocks) and fadvised altogether.
+ */
+static void
+prefetch_next_incremental_run(int fd, BlockNumber *incremental_blocks,
+							  unsigned num_incremental_blocks,
+							  unsigned max_run_len,
+							  unsigned *prefetch_idx)
+{
+	BlockNumber run_start = incremental_blocks[(*prefetch_idx)++];
+	unsigned	run_len = 1;
+
+	/*
+	 * Merge contiguous blocks into a single run, up to max_run_len
+	 */
+	while (run_len < max_run_len &&
+		   *prefetch_idx < num_incremental_blocks &&
+		   incremental_blocks[*prefetch_idx] == run_start + run_len)
+	{
+		run_len++;
+		(*prefetch_idx)++;
+	}
+
+	(void) posix_fadvise(fd, (off_t) run_start * BLCKSZ,
+						 (off_t) run_len * BLCKSZ,
+						 POSIX_FADV_WILLNEED);
+}
+#endif
+
 /*
  * Read some more data from the file into the bbsink's buffer, verifying
  * checksums as required.
-- 
2.43.0

