From 7712c20f11a9273e74a51ec3bb4c8775132935c5 Mon Sep 17 00:00:00 2001 From: Peter Geoghegan Date: Tue, 25 Nov 2025 18:03:15 -0500 Subject: [PATCH v37 03/11] Adopt amgetbatch interface in hash index AM. Replace hashgettuple with hashgetbatch, a function that implements the new amgetbatch interface added by commit FIXME. Plain index scans of hash indexes now return matching items in batches consisting of all of the matches from a given bucket page or overflow page. This gives the table AM the ability to perform optimizations like index prefetching during hash index scans. The amgetbatch interface requires that index AMs take the same standardized approach to pin management for pins that are used to prevent unsafe concurrent TID recycling by VACUUM (that way prefetching can hold open multiple batches without it affecting the read stream). Note, however, that hash still keeps the pins it needs for its own purposes, on the primary page of each bucket the scan reads. They are now held until the scan is restarted or ended, since the table AM may step backward from the last batch returned. Under amgettuple, hash index scans stepped between pages in two places. When a page had no matching items, _hash_readpage went on to the next page itself, through _hash_readnext and _hash_readprev, which also knew how to cross from one bucket of an in-progress split to the other. When a page did have matching items, _hash_next stepped later, from block numbers that _hash_readpage had saved, with a second copy of that logic. Which copy ran depended on whether the page being left had any matches, which has nothing to do with the structure of the index; the bug fixed by commit 173cf85a was that only the first copy knew about the second bucket case. Now _hash_readpage only ever reads a single page per call, and leaves the page's lock and pin alone, so stepping between pages, and across the buckets of a split, happens in a single new routine, _hash_readnextpage. This follows nbtree: _hash_readnextpage is modeled on _bt_readnextpage, and _hash_readpage now works the way _bt_readpage always has. hashkillitemsbatch (the hash implementation of the new amkillitemsbatch interface) performs LP_DEAD marking of dead index entries in the standard way for amgetbatch routines. Preparatory commit e5836f7b added support for fake LSNs to the hash index AM, so that hashkillitemsbatch's LSN test works just as well during scans of unlogged indexes. Author: Peter Geoghegan Reviewed-by: Tomas Vondra Reviewed-by: Andres Freund Discussion: https://postgr.es/m/CAH2-WzmYqhacBH161peAWb5eF=Ja7CFAQ+0jSEMq=qnfLVTOOg@mail.gmail.com --- src/include/access/hash.h | 96 +--- src/backend/access/hash/README | 58 ++- src/backend/access/hash/hash.c | 238 ++++++--- src/backend/access/hash/hash_xlog.c | 4 +- src/backend/access/hash/hashpage.c | 22 +- src/backend/access/hash/hashsearch.c | 750 +++++++++++---------------- src/backend/access/hash/hashutil.c | 128 +---- src/backend/access/index/batchscan.c | 2 +- doc/src/sgml/indexam.sgml | 32 +- src/tools/pgindent/typedefs.list | 3 +- 10 files changed, 545 insertions(+), 788 deletions(-) diff --git a/src/include/access/hash.h b/src/include/access/hash.h index a8702f0e5..bfcfe5814 100644 --- a/src/include/access/hash.h +++ b/src/include/access/hash.h @@ -100,58 +100,6 @@ typedef HashPageOpaqueData *HashPageOpaque; */ #define HASHO_PAGE_ID 0xFF80 -typedef struct HashScanPosItem /* what we remember about each match */ -{ - ItemPointerData heapTid; /* TID of referenced heap item */ - OffsetNumber indexOffset; /* index item's location within page */ -} HashScanPosItem; - -typedef struct HashScanPosData -{ - Buffer buf; /* if valid, the buffer is pinned */ - BlockNumber currPage; /* current hash index page */ - BlockNumber nextPage; /* next overflow page */ - BlockNumber prevPage; /* prev overflow or bucket page */ - - /* - * The items array is always ordered in index order (ie, increasing - * indexoffset). When scanning backwards it is convenient to fill the - * array back-to-front, so we start at the last slot and fill downwards. - * Hence we need both a first-valid-entry and a last-valid-entry counter. - * itemIndex is a cursor showing which entry was last returned to caller. - */ - int firstItem; /* first valid index in items[] */ - int lastItem; /* last valid index in items[] */ - int itemIndex; /* current index in items[] */ - - HashScanPosItem items[MaxIndexTuplesPerPage]; /* MUST BE LAST */ -} HashScanPosData; - -#define HashScanPosIsPinned(scanpos) \ -( \ - AssertMacro(BlockNumberIsValid((scanpos).currPage) || \ - !BufferIsValid((scanpos).buf)), \ - BufferIsValid((scanpos).buf) \ -) - -#define HashScanPosIsValid(scanpos) \ -( \ - AssertMacro(BlockNumberIsValid((scanpos).currPage) || \ - !BufferIsValid((scanpos).buf)), \ - BlockNumberIsValid((scanpos).currPage) \ -) - -#define HashScanPosInvalidate(scanpos) \ - do { \ - (scanpos).buf = InvalidBuffer; \ - (scanpos).currPage = InvalidBlockNumber; \ - (scanpos).nextPage = InvalidBlockNumber; \ - (scanpos).prevPage = InvalidBlockNumber; \ - (scanpos).firstItem = 0; \ - (scanpos).lastItem = 0; \ - (scanpos).itemIndex = 0; \ - } while (0) - /* * HashScanOpaqueData is private state for a hash index scan. */ @@ -172,25 +120,27 @@ typedef struct HashScanOpaqueData /* Whether scan starts on bucket being populated due to split */ bool hashso_buc_populated; - - /* - * Whether scanning bucket being split? The value of this parameter is - * referred only when hashso_buc_populated is true. - */ - bool hashso_buc_split; - /* info about killed items if any (killedItems is NULL if never used) */ - int *killedItems; /* currPos.items indexes of killed items */ - int numKilled; /* number of currently stored items */ - - /* - * Identify all the matching items on a page and save them in - * HashScanPosData - */ - HashScanPosData currPos; /* current position data */ } HashScanOpaqueData; typedef HashScanOpaqueData *HashScanOpaque; +/* Per-batch data private to the hash index AM */ +typedef struct HashBatchData +{ + Buffer buf; /* index page's buffer pin */ + BlockNumber batchPage; /* index page's block number */ + BlockNumber prevPage; /* index page's left sibling + * (InvalidBlockNumber for bucket pages, whose + * hasho_prevblkno isn't a real block number) */ + BlockNumber nextPage; /* index page's right sibling */ + bool bucSplit; /* index page in bucket being split? (only + * possible when hashso_buc_populated) */ +} HashBatchData; + +/* Access the hash-private per-batch data from an IndexScanBatch pointer */ +#define HashBatchGetData(scan, batch) \ + index_scan_batch_index_opaque_static(scan, batch, HashBatchData) + /* * Definitions for metapage. */ @@ -368,11 +318,15 @@ extern bool hashinsert(Relation rel, Datum *values, bool *isnull, IndexUniqueCheck checkUnique, bool indexUnchanged, struct IndexInfo *indexInfo); -extern bool hashgettuple(IndexScanDesc scan, ScanDirection dir); +extern IndexScanBatch hashgetbatch(IndexScanDesc scan, + IndexScanBatch priorbatch, + ScanDirection dir); extern int64 hashgetbitmap(IndexScanDesc scan, TIDBitmap *tbm); extern IndexScanDesc hashbeginscan(Relation rel, int nkeys, int norderbys); extern void hashrescan(IndexScanDesc scan, ScanKey scankey, int nscankeys, ScanKey orderbys, int norderbys); +extern void hashunguardbatch(IndexScanDesc scan, IndexScanBatch batch); +extern void hashkillitemsbatch(IndexScanDesc scan, IndexScanBatch batch); extern void hashendscan(IndexScanDesc scan); extern IndexBulkDeleteResult *hashbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats, @@ -445,8 +399,9 @@ extern void _hash_finish_split(Relation rel, Buffer metabuf, Buffer obuf, uint32 lowmask); /* hashsearch.c */ -extern bool _hash_next(IndexScanDesc scan, ScanDirection dir); -extern bool _hash_first(IndexScanDesc scan, ScanDirection dir); +extern IndexScanBatch _hash_next(IndexScanDesc scan, ScanDirection dir, + IndexScanBatch priorbatch); +extern IndexScanBatch _hash_first(IndexScanDesc scan, ScanDirection dir); /* hashsort.c */ typedef struct HSpool HSpool; /* opaque struct in hashsort.c */ @@ -476,7 +431,6 @@ extern BlockNumber _hash_get_oldblock_from_newbucket(Relation rel, Bucket new_bu extern BlockNumber _hash_get_newblock_from_oldbucket(Relation rel, Bucket old_bucket); extern Bucket _hash_get_newbucket_from_oldbucket(Relation rel, Bucket old_bucket, uint32 lowmask, uint32 maxbucket); -extern void _hash_kill_items(IndexScanDesc scan); /* hash.c */ extern void hashbucketcleanup(Relation rel, Bucket cur_bucket, diff --git a/src/backend/access/hash/README b/src/backend/access/hash/README index fc9031117..61789f7cf 100644 --- a/src/backend/access/hash/README +++ b/src/backend/access/hash/README @@ -255,34 +255,44 @@ The reader algorithm is: retake the buffer content lock on new bucket arrange to scan the old bucket normally and the new bucket for tuples which are not moved-by-split --- then, per read request: - reacquire content lock on current page - step to next page if necessary (no chaining of content locks, but keep - the pin on the primary bucket throughout the scan) - save all the matching tuples from current index page into an items array - release pin and content lock (but if it is primary bucket page retain - its pin till the end of the scan) - get tuple from an item array --- at scan shutdown: - release all pins still held +-- then, per batch (page) request: + step to the next page in the scan direction, by the block number saved + with the previous batch; at the end of a bucket's chain, a scan that + started during a split crosses to the other bucket of the split (no + chaining of content locks, but keep the pin on the primary bucket + throughout the scan) + pin and lock that page, and save all the matching tuples from it into a + batch (a page with no matching tuples is released and passed over) + release content lock on current page, then return batch to table AM + (table AM will drop batch's buffer pin, though primary bucket page pin + is kept until the end of the scan) +-- when the scan ends or restarts: + release scan-owned pins (the primary bucket page pins); these outlive a + batch request that finds no further matches, because the table AM may + still reverse direction from a batch it holds Holding the buffer pin on the primary bucket page for the whole scan prevents -the reader's current-tuple pointer from being invalidated by splits or -compactions. (Of course, other buckets can still be split or compacted.) +the bucket from being reorganized by splits or compactions while the scan is +in progress. (Of course, other buckets can still be split or compacted.) -To minimize lock/unlock traffic, hash index scan always searches the entire -hash page to identify all the matching items at once, copying their heap tuple -IDs into backend-local storage. The heap tuple IDs are then processed while not -holding any page lock within the index thereby, allowing concurrent insertion -to happen on the same index page without any requirement of re-finding the -current scan position for the reader. We do continue to hold a pin on the -bucket page, to protect against concurrent deletions and bucket split. +To minimize lock/unlock traffic, hash index scans always search the entire +hash page to identify all the matching items at once, returning them in +batches to the table AM. The table AM processes batches while no page lock +is held within the index, allowing concurrent insertion to happen on the +same index page without any requirement of re-finding the current scan +position for the reader. The table AM controls when batch buffer pins are +dropped. We do continue to hold a pin on the primary bucket page, to +protect against concurrent bucket splits. -To allow for scans during a bucket split, if at the start of the scan, the -bucket is marked as bucket-being-populated, it scan all the tuples in that -bucket except for those that are marked as moved-by-split. Once it finishes -the scan of all the tuples in the current bucket, it scans the old bucket from -which this bucket is formed by split. +To allow for scans during a bucket split, if at the start of the scan the +bucket is marked as bucket-being-populated, the scan reads both buckets of +the split: the bucket being populated, skipping the tuples that are marked as +moved-by-split, and the old bucket from which it is being formed, which still +holds the originals of those tuples. A forward scan reads the bucket being +populated first and then the old bucket. A backward scan reads the same +pages in the reverse order, starting at the end of the old bucket's chain, so +that a scrollable cursor sees one consistent order. The scan keeps a pin on +the primary page of both buckets until it ends. The insertion algorithm is rather similar: diff --git a/src/backend/access/hash/hash.c b/src/backend/access/hash/hash.c index 20b70e538..b80825c5e 100644 --- a/src/backend/access/hash/hash.c +++ b/src/backend/access/hash/hash.c @@ -114,10 +114,10 @@ hashhandler(PG_FUNCTION_ARGS) .amadjustmembers = hashadjustmembers, .ambeginscan = hashbeginscan, .amrescan = hashrescan, - .amgettuple = hashgettuple, - .amgetbatch = NULL, - .amunguardbatch = NULL, - .amkillitemsbatch = NULL, + .amgettuple = NULL, + .amgetbatch = hashgetbatch, + .amunguardbatch = hashunguardbatch, + .amkillitemsbatch = hashkillitemsbatch, .amgetbitmap = hashgetbitmap, .amendscan = hashendscan, .amposreset = NULL, @@ -300,53 +300,51 @@ hashinsert(Relation rel, Datum *values, bool *isnull, /* - * hashgettuple() -- Get the next tuple in the scan. + * hashgetbatch() -- Get the first or next batch of tuples in the scan */ -bool -hashgettuple(IndexScanDesc scan, ScanDirection dir) +IndexScanBatch +hashgetbatch(IndexScanDesc scan, IndexScanBatch priorbatch, ScanDirection dir) { HashScanOpaque so = (HashScanOpaque) scan->opaque; - bool res; + IndexScanBatch batch; /* Hash indexes are always lossy since we store only the hash code */ - scan->xs_recheck = true; + Assert(scan->xs_recheck); - /* - * If we've already initialized this scan, we can just advance it in the - * appropriate direction. If we haven't done so yet, we call a routine to - * get the first item in the scan. - */ - if (!HashScanPosIsValid(so->currPos)) - res = _hash_first(scan, dir); + if (priorbatch == NULL) + { + _hash_dropscanbuf(scan->indexRelation, so); + + /* Initialize the scan, and get the first batch of matching items */ + batch = _hash_first(scan, dir); + } else { - /* - * Check to see if we should kill the previously-fetched tuple. - */ - if (scan->kill_prior_tuple) - { - /* - * Yes, so remember it for later. (We'll deal with all such tuples - * at once right after leaving the index page or at end of scan.) - * In case if caller reverses the indexscan direction it is quite - * possible that the same item might get entered multiple times. - * But, we don't detect that; instead, we just forget any excess - * entries. - */ - if (so->killedItems == NULL) - so->killedItems = palloc_array(int, MaxIndexTuplesPerPage); - - if (so->numKilled < MaxIndexTuplesPerPage) - so->killedItems[so->numKilled++] = so->currPos.itemIndex; - } - - /* - * Now continue the scan. - */ - res = _hash_next(scan, dir); + /* Get batch positioned after caller's batch (in direction 'dir') */ + batch = _hash_next(scan, dir, priorbatch); } - return res; + /* + * When the batch's page has no next/previous page for _hash_next to step + * to in a direction, the scan already ended on the returned batch; mark + * the batch accordingly. A scan that started during a bucket split steps + * from the end of the first bucket it reads to the other bucket: forward + * scans only end in the bucket being split, backward scans only in the + * bucket being populated. + */ + if (batch) + { + HashBatchData *hashbatch = HashBatchGetData(scan, batch); + + if (!BlockNumberIsValid(hashbatch->nextPage) && + (!so->hashso_buc_populated || hashbatch->bucSplit)) + batch->knownEndForward = true; + if (!BlockNumberIsValid(hashbatch->prevPage) && + (!so->hashso_buc_populated || !hashbatch->bucSplit)) + batch->knownEndBackward = true; + } + + return batch; } @@ -356,26 +354,26 @@ hashgettuple(IndexScanDesc scan, ScanDirection dir) int64 hashgetbitmap(IndexScanDesc scan, TIDBitmap *tbm) { - HashScanOpaque so = (HashScanOpaque) scan->opaque; - bool res; + IndexScanBatch batch; int64 ntids = 0; - HashScanPosItem *currItem; - res = _hash_first(scan, ForwardScanDirection); + batch = _hash_first(scan, ForwardScanDirection); - while (res) + while (batch != NULL) { - currItem = &so->currPos.items[so->currPos.itemIndex]; + for (int itemIndex = batch->firstItem; + itemIndex <= batch->lastItem; + itemIndex++) + { + tbm_add_tuples(tbm, &batch->items[itemIndex].tableTid, 1, true); + ntids++; + } /* - * _hash_first and _hash_next handle eliminate dead index entries - * whenever scan->ignore_killed_tuples is true. Therefore, there's - * nothing to do here except add the results to the TIDBitmap. + * _hash_next releases the prior batch for bitmap callers before + * allocating the next one, so only one batch is ever used at a time */ - tbm_add_tuples(tbm, &(currItem->heapTid), 1, true); - ntids++; - - res = _hash_next(scan, ForwardScanDirection); + batch = _hash_next(scan, ForwardScanDirection, batch); } return ntids; @@ -397,17 +395,16 @@ hashbeginscan(Relation rel, int nkeys, int norderbys) scan = RelationGetIndexScan(rel, nkeys, norderbys); so = (HashScanOpaque) palloc_object(HashScanOpaqueData); - HashScanPosInvalidate(so->currPos); so->hashso_bucket_buf = InvalidBuffer; so->hashso_split_bucket_buf = InvalidBuffer; so->hashso_buc_populated = false; - so->hashso_buc_split = false; - - so->killedItems = NULL; - so->numKilled = 0; scan->opaque = so; + scan->xs_recheck = true; + scan->maxitemsbatch = MaxIndexTuplesPerPage; + scan->batch_index_opaque_static = MAXALIGN(sizeof(HashBatchData)); + scan->batch_tuples_workspace = 0; return scan; } @@ -422,24 +419,116 @@ hashrescan(IndexScanDesc scan, ScanKey scankey, int nscankeys, HashScanOpaque so = (HashScanOpaque) scan->opaque; Relation rel = scan->indexRelation; - if (HashScanPosIsValid(so->currPos)) - { - /* Before leaving current page, deal with any killed items */ - if (so->numKilled > 0) - _hash_kill_items(scan); - } - _hash_dropscanbuf(rel, so); - /* set position invalid (this will cause _hash_first call) */ - HashScanPosInvalidate(so->currPos); - /* Update scan key, if a new one is given */ if (scankey && scan->numberOfKeys > 0) memcpy(scan->keyData, scankey, scan->numberOfKeys * sizeof(ScanKeyData)); so->hashso_buc_populated = false; - so->hashso_buc_split = false; +} + +/* + * hashunguardbatch() -- Drop batch's TID recycling interlock (buffer pin) + * + * Called by the table AM when it's safe to drop the buffer pin held to + * prevent concurrent TID recycling by VACUUM. + */ +void +hashunguardbatch(IndexScanDesc scan, IndexScanBatch batch) +{ + HashBatchData *hashbatch = HashBatchGetData(scan, batch); + + /* Should be called exactly once iff !batchImmediateUnguard */ + Assert(!scan->batchImmediateUnguard); + Assert(batch->isGuarded); + Assert(BufferGetBlockNumber(hashbatch->buf) == hashbatch->batchPage); + + ReleaseBuffer(hashbatch->buf); +} + +/* + * hashkillitemsbatch() -- Mark dead items' index tuples LP_DEAD + */ +void +hashkillitemsbatch(IndexScanDesc scan, IndexScanBatch batch) +{ + Relation rel = scan->indexRelation; + HashBatchData *hashbatch = HashBatchGetData(scan, batch); + Buffer buf; + Page page; + HashPageOpaque opaque; + bool killedsomething = false; + XLogRecPtr latestlsn; + + Assert(batch->numDead > 0); + + buf = _hash_getbuf(rel, hashbatch->batchPage, HASH_READ, + LH_BUCKET_PAGE | LH_OVERFLOW_PAGE); + + latestlsn = BufferGetLSNAtomic(buf); + Assert(batch->lsn <= latestlsn); + if (batch->lsn != latestlsn) + { + /* Modified, give up on hinting */ + _hash_relbuf(rel, buf); + return; + } + + page = BufferGetPage(buf); + opaque = HashPageGetOpaque(page); + + /* Iterate through batch->deadItems[] in index page order */ + for (int i = 0; i < batch->numDead; i++) + { + int itemIndex = batch->deadItems[i]; + BatchMatchingItem *currItem = &batch->items[itemIndex]; + OffsetNumber offnum = currItem->indexOffset; + ItemId iid = PageGetItemId(page, offnum); + + Assert(itemIndex >= batch->firstItem && + itemIndex <= batch->lastItem); + Assert(i == 0 || + offnum > batch->items[batch->deadItems[i - 1]].indexOffset); + Assert(offnum <= PageGetMaxOffsetNumber(page)); + Assert(ItemPointerEquals(&((IndexTuple) PageGetItem(page, iid))->t_tid, + &currItem->tableTid)); + + /* Mark index item as dead, if it isn't already */ + if (!ItemIdIsDead(iid)) + { + if (!killedsomething) + { + /* + * Use the hint bit infrastructure to check if we can update + * the page while just holding a share lock. If we are not + * allowed, there's no point continuing. + */ + if (!BufferBeginSetHintBits(buf)) + { + _hash_relbuf(rel, buf); + return; + } + } + + /* found the item */ + ItemIdMarkDead(iid); + killedsomething = true; + } + } + + /* + * Since this can be redone later if needed, mark as dirty hint. Whenever + * we mark anything LP_DEAD, we also set the page's + * LH_PAGE_HAS_DEAD_TUPLES flag, which is likewise just a hint. + */ + if (killedsomething) + { + opaque->hasho_flag |= LH_PAGE_HAS_DEAD_TUPLES; + BufferFinishSetHintBits(buf, true, true); + } + + _hash_relbuf(rel, buf); } /* @@ -451,17 +540,8 @@ hashendscan(IndexScanDesc scan) HashScanOpaque so = (HashScanOpaque) scan->opaque; Relation rel = scan->indexRelation; - if (HashScanPosIsValid(so->currPos)) - { - /* Before leaving current page, deal with any killed items */ - if (so->numKilled > 0) - _hash_kill_items(scan); - } - _hash_dropscanbuf(rel, so); - if (so->killedItems != NULL) - pfree(so->killedItems); pfree(so); scan->opaque = NULL; } diff --git a/src/backend/access/hash/hash_xlog.c b/src/backend/access/hash/hash_xlog.c index e9a2b9aa9..a0292c2ae 100644 --- a/src/backend/access/hash/hash_xlog.c +++ b/src/backend/access/hash/hash_xlog.c @@ -1118,14 +1118,14 @@ hash_mask(char *pagedata, BlockNumber blkno) /* * In hash bucket and overflow pages, it is possible to modify the * LP_FLAGS without emitting any WAL record. Hence, mask the line - * pointer flags. See hashgettuple(), _hash_kill_items() for details. + * pointer flags. See hashkillitemsbatch() for details. */ mask_lp_flags(page); } /* * It is possible that the hint bit LH_PAGE_HAS_DEAD_TUPLES may remain - * unlogged. So, mask it. See _hash_kill_items() for details. + * unlogged. So, mask it. See hashkillitemsbatch() for details. */ opaque->hasho_flag &= ~LH_PAGE_HAS_DEAD_TUPLES; } diff --git a/src/backend/access/hash/hashpage.c b/src/backend/access/hash/hashpage.c index fc11ed6b7..40c928de9 100644 --- a/src/backend/access/hash/hashpage.c +++ b/src/backend/access/hash/hashpage.c @@ -281,34 +281,26 @@ _hash_dropbuf(Relation rel, Buffer buf) } /* - * _hash_dropscanbuf() -- release buffers used in scan. + * _hash_dropscanbuf() -- release buffers owned by scan. * - * This routine unpins the buffers used during scan on which we - * hold no lock. + * This routine unpins the buffers for the primary bucket page and for the + * bucket page of a bucket being split as needed. */ void _hash_dropscanbuf(Relation rel, HashScanOpaque so) { - /* release pin we hold on primary bucket page */ - if (BufferIsValid(so->hashso_bucket_buf) && - so->hashso_bucket_buf != so->currPos.buf) + /* release pin held on primary bucket page */ + if (BufferIsValid(so->hashso_bucket_buf)) _hash_dropbuf(rel, so->hashso_bucket_buf); so->hashso_bucket_buf = InvalidBuffer; - /* release pin we hold on primary bucket page of bucket being split */ - if (BufferIsValid(so->hashso_split_bucket_buf) && - so->hashso_split_bucket_buf != so->currPos.buf) + /* release pin held on primary bucket page of bucket being split */ + if (BufferIsValid(so->hashso_split_bucket_buf)) _hash_dropbuf(rel, so->hashso_split_bucket_buf); so->hashso_split_bucket_buf = InvalidBuffer; - /* release any pin we still hold */ - if (BufferIsValid(so->currPos.buf)) - _hash_dropbuf(rel, so->currPos.buf); - so->currPos.buf = InvalidBuffer; - /* reset split scan */ so->hashso_buc_populated = false; - so->hashso_buc_split = false; } diff --git a/src/backend/access/hash/hashsearch.c b/src/backend/access/hash/hashsearch.c index 46eb12cc8..59b54ffd3 100644 --- a/src/backend/access/hash/hashsearch.c +++ b/src/backend/access/hash/hashsearch.c @@ -14,6 +14,7 @@ */ #include "postgres.h" +#include "access/batchscan.h" #include "access/hash.h" #include "access/relscan.h" #include "miscadmin.h" @@ -22,253 +23,214 @@ #include "storage/predicate.h" #include "utils/rel.h" -static void _hash_readnext(IndexScanDesc scan, Buffer *bufp, - Page *pagep, HashPageOpaque *opaquep); -static void _hash_readprev(IndexScanDesc scan, Buffer *bufp, - Page *pagep, HashPageOpaque *opaquep); +static IndexScanBatch _hash_readfirstpage(IndexScanDesc scan, + IndexScanBatch firstbatch, Buffer buf, + ScanDirection dir); +static IndexScanBatch _hash_readnextpage(IndexScanDesc scan, BlockNumber blkno, + ScanDirection dir, bool bucSplit); static Buffer _hash_step_to_split_bucket(IndexScanDesc scan); static Buffer _hash_step_to_populated_bucket(IndexScanDesc scan); -static bool _hash_readpage(IndexScanDesc scan, Buffer *bufP, - ScanDirection dir); -static int _hash_load_qualified_items(IndexScanDesc scan, Page page, - OffsetNumber offnum, ScanDirection dir); -static inline void _hash_saveitem(HashScanOpaque so, int itemIndex, +static Buffer _hash_chain_end(IndexScanDesc scan, Buffer buf); +static bool _hash_readpage(IndexScanDesc scan, Buffer buf, ScanDirection dir, + IndexScanBatch batch, bool bucSplit); +static inline void _hash_saveitem(IndexScanBatch batch, int itemIndex, OffsetNumber offnum, IndexTuple itup); /* - * _hash_next() -- Get the next item in a scan. + * _hash_next() -- Get the next batch of items in a scan. * - * On entry, so->currPos describes the current page, which may - * be pinned but not locked, and so->currPos.itemIndex identifies - * which item was previously returned. + * On entry, priorbatch describes the current page batch with items + * already returned. * - * On successful exit, scan->xs_heaptid is set to the TID of the next - * heap tuple. so->currPos is updated as needed. + * On successful exit, returns a batch containing matching items from + * the next page that has any. Otherwise returns NULL, indicating that + * there are no further matches. No locks are ever held when we return. * - * On failure exit (no more tuples), we return false with pin - * held on bucket page but no pins or locks held on overflow - * page. + * Retains pins according to the same rules as _hash_first. */ -bool -_hash_next(IndexScanDesc scan, ScanDirection dir) +IndexScanBatch +_hash_next(IndexScanDesc scan, ScanDirection dir, IndexScanBatch priorbatch) +{ + HashBatchData *hashpriorbatch = HashBatchGetData(scan, priorbatch); + BlockNumber blkno; + bool bucSplit; + + /* + * The core code must deal with cross-batch scan direction changes for us. + * A batch management routine that flips priorbatch's scan direction is + * used for this. + */ + Assert(priorbatch->dir == dir); + + /* Step from priorbatch's page to its neighbor in this scan direction */ + if (ScanDirectionIsForward(dir)) + blkno = hashpriorbatch->nextPage; + else + blkno = hashpriorbatch->prevPage; + bucSplit = hashpriorbatch->bucSplit; + + /* + * For bitmap scan callers, release the prior batch now so that + * _hash_readnextpage can reuse its memory. That way bitmap scans never + * need more than one batch allocation. + */ + if (!scan->usebatchring) + batchscan_release(scan, priorbatch); + + return _hash_readnextpage(scan, blkno, dir, bucSplit); +} + +/* + * _hash_readfirstpage() -- Read the first page of a scan, for _hash_first. + * + * firstbatch is the batch allocated for the first page's matches. buf + * is the primary page of the bucket that the scan key maps to, + * share-locked and pinned for us (scan maintains its own separate pin). + * A forward scan reads it first. A backward scan reads bucket chains + * tail to head, so it starts at the last page of the chain; if it + * started during a bucket split, it reads the bucket being split first, + * so it starts at the end of that bucket's chain instead. + * + * Returns a batch containing matching items, or NULL at the end of the + * scan. No locks are ever held when we return. + */ +static IndexScanBatch +_hash_readfirstpage(IndexScanDesc scan, IndexScanBatch firstbatch, Buffer buf, + ScanDirection dir) { Relation rel = scan->indexRelation; HashScanOpaque so = (HashScanOpaque) scan->opaque; - HashScanPosItem *currItem; + HashBatchData *hashfirstbatch; BlockNumber blkno; - Buffer buf; - bool end_of_scan = false; + bool bucSplit = false; + + if (ScanDirectionIsBackward(dir)) + { + if (so->hashso_buc_populated) + { + _hash_relbuf(rel, buf); + buf = _hash_step_to_split_bucket(scan); + bucSplit = true; + } + buf = _hash_chain_end(scan, buf); + } + + if (_hash_readpage(scan, buf, dir, firstbatch, bucSplit)) + { + /* _hash_readpage saved one or more matches in firstbatch.items[] */ + batchscan_unlock(scan, firstbatch, buf); + return firstbatch; + } /* - * Advance to the next tuple on the current page; or if done, try to read - * data from the next or previous page based on the scan direction. Before - * moving to the next or previous page make sure that we deal with all the - * killed items. + * No matching items on the first page. Go on from its neighbor in the + * scan direction, after releasing the page and firstbatch + * (_hash_readnextpage will recycle the batch). */ + _hash_relbuf(rel, buf); + hashfirstbatch = HashBatchGetData(scan, firstbatch); if (ScanDirectionIsForward(dir)) + blkno = hashfirstbatch->nextPage; + else + blkno = hashfirstbatch->prevPage; + batchscan_release(scan, firstbatch); + + return _hash_readnextpage(scan, blkno, dir, bucSplit); +} + +/* + * _hash_readnextpage() -- Read the next page with matching items. + * + * blkno is the next page in the scan direction, or InvalidBlockNumber + * when the page the scan is leaving has no neighbor in that direction. + * bucSplit says whether that page is in the bucket being split, which + * only matters when the scan started during a split: such a scan reads + * both buckets of the split, so at the end of a chain it crosses to the + * other bucket, a forward scan from the end of the bucket being + * populated to the start of the bucket being split, a backward scan + * the other way around. We read pages in the scan direction until one + * has a matching item, or until the scan runs out of pages. + * + * On entry, no page is locked. Returns a batch containing matching + * items, or NULL at the end of the scan. No locks are ever held when + * we return. + */ +static IndexScanBatch +_hash_readnextpage(IndexScanDesc scan, BlockNumber blkno, ScanDirection dir, + bool bucSplit) +{ + Relation rel = scan->indexRelation; + HashScanOpaque so = (HashScanOpaque) scan->opaque; + IndexScanBatch newbatch; + HashBatchData *hashnewbatch; + Buffer buf; + + /* only a scan that started during a split can be in either bucket */ + Assert(!bucSplit || so->hashso_buc_populated); + + /* Allocate space for new batch before locking anything */ + newbatch = batchscan_alloc(scan); + hashnewbatch = HashBatchGetData(scan, newbatch); + + for (;;) { - if (++so->currPos.itemIndex > so->currPos.lastItem) + /* check for interrupts while we're not holding any buffer lock */ + CHECK_FOR_INTERRUPTS(); + + if (ScanDirectionIsForward(dir)) { - if (so->numKilled > 0) - _hash_kill_items(scan); - - blkno = so->currPos.nextPage; if (BlockNumberIsValid(blkno)) - { buf = _hash_getbuf(rel, blkno, HASH_READ, LH_OVERFLOW_PAGE); - if (!_hash_readpage(scan, &buf, dir)) - end_of_scan = true; - } - else if (so->hashso_buc_populated && !so->hashso_buc_split) + else if (so->hashso_buc_populated && !bucSplit) { + /* Switch from populated bucket to split bucket */ buf = _hash_step_to_split_bucket(scan); - - if (!_hash_readpage(scan, &buf, dir)) - end_of_scan = true; + bucSplit = true; } else - end_of_scan = true; + break; } - } - else - { - if (--so->currPos.itemIndex < so->currPos.firstItem) + else { - if (so->numKilled > 0) - _hash_kill_items(scan); - - blkno = so->currPos.prevPage; if (BlockNumberIsValid(blkno)) - { buf = _hash_getbuf(rel, blkno, HASH_READ, LH_BUCKET_PAGE | LH_OVERFLOW_PAGE); - - /* - * We always maintain the pin on bucket page for whole scan - * operation, so releasing the additional pin we have acquired - * here. - */ - if (buf == so->hashso_bucket_buf || - buf == so->hashso_split_bucket_buf) - _hash_dropbuf(rel, buf); - - if (!_hash_readpage(scan, &buf, dir)) - end_of_scan = true; - } - else if (so->hashso_buc_populated && so->hashso_buc_split) + else if (so->hashso_buc_populated && bucSplit) { + /* Switch from split bucket to populated bucket */ buf = _hash_step_to_populated_bucket(scan); - - if (!_hash_readpage(scan, &buf, dir)) - end_of_scan = true; + bucSplit = false; } else - end_of_scan = true; + break; } + + if (_hash_readpage(scan, buf, dir, newbatch, bucSplit)) + { + /* _hash_readpage saved one or more matches in newbatch.items[] */ + batchscan_unlock(scan, newbatch, buf); + return newbatch; + } + + /* No matching items on that page; release it, go on from its neighbor */ + _hash_relbuf(rel, buf); + if (ScanDirectionIsForward(dir)) + blkno = hashnewbatch->nextPage; + else + blkno = hashnewbatch->prevPage; } - if (end_of_scan) - { - /* - * A scan that started during a bucket split ends in the bucket that - * its direction visits last: forward scans end in the bucket being - * split, backward scans in the bucket being populated. - */ - Assert(!so->hashso_buc_populated || - so->hashso_buc_split == ScanDirectionIsForward(dir)); + batchscan_release(scan, newbatch); - _hash_dropscanbuf(rel, so); - HashScanPosInvalidate(so->currPos); - return false; - } - - /* OK, itemIndex says what to return */ - currItem = &so->currPos.items[so->currPos.itemIndex]; - scan->xs_heaptid = currItem->heapTid; - - return true; -} - -/* - * Advance to next page in a bucket, if any. If we are scanning the bucket - * being populated during split operation then this function advances to the - * bucket being split after the last bucket page of bucket being populated. - */ -static void -_hash_readnext(IndexScanDesc scan, - Buffer *bufp, Page *pagep, HashPageOpaque *opaquep) -{ - BlockNumber blkno; - Relation rel = scan->indexRelation; - HashScanOpaque so = (HashScanOpaque) scan->opaque; - bool block_found = false; - - blkno = (*opaquep)->hasho_nextblkno; - - /* - * Retain the pin on primary bucket page till the end of scan. Refer the - * comments in _hash_first to know the reason of retaining pin. - */ - if (*bufp == so->hashso_bucket_buf || *bufp == so->hashso_split_bucket_buf) - LockBuffer(*bufp, BUFFER_LOCK_UNLOCK); - else - _hash_relbuf(rel, *bufp); - - *bufp = InvalidBuffer; - /* check for interrupts while we're not holding any buffer lock */ - CHECK_FOR_INTERRUPTS(); - if (BlockNumberIsValid(blkno)) - { - *bufp = _hash_getbuf(rel, blkno, HASH_READ, LH_OVERFLOW_PAGE); - block_found = true; - } - else if (so->hashso_buc_populated && !so->hashso_buc_split) - { - /* - * end of bucket, scan bucket being split if there was a split in - * progress at the start of scan. - */ - *bufp = _hash_step_to_split_bucket(scan); - block_found = true; - } - - if (block_found) - { - *pagep = BufferGetPage(*bufp); - *opaquep = HashPageGetOpaque(*pagep); - } -} - -/* - * Advance to previous page in a bucket, if any. If the current scan has - * started during split operation then this function advances to bucket - * being populated after the first bucket page of bucket being split. - */ -static void -_hash_readprev(IndexScanDesc scan, - Buffer *bufp, Page *pagep, HashPageOpaque *opaquep) -{ - BlockNumber blkno; - Relation rel = scan->indexRelation; - HashScanOpaque so = (HashScanOpaque) scan->opaque; - bool haveprevblk; - - blkno = (*opaquep)->hasho_prevblkno; - - /* - * Retain the pin on primary bucket page till the end of scan. Refer the - * comments in _hash_first to know the reason of retaining pin. - */ - if (*bufp == so->hashso_bucket_buf || *bufp == so->hashso_split_bucket_buf) - { - LockBuffer(*bufp, BUFFER_LOCK_UNLOCK); - haveprevblk = false; - } - else - { - _hash_relbuf(rel, *bufp); - haveprevblk = true; - } - - *bufp = InvalidBuffer; - /* check for interrupts while we're not holding any buffer lock */ - CHECK_FOR_INTERRUPTS(); - - if (haveprevblk) - { - Assert(BlockNumberIsValid(blkno)); - *bufp = _hash_getbuf(rel, blkno, HASH_READ, - LH_BUCKET_PAGE | LH_OVERFLOW_PAGE); - *pagep = BufferGetPage(*bufp); - *opaquep = HashPageGetOpaque(*pagep); - - /* - * We always maintain the pin on bucket page for whole scan operation, - * so releasing the additional pin we have acquired here. - */ - if (*bufp == so->hashso_bucket_buf || *bufp == so->hashso_split_bucket_buf) - _hash_dropbuf(rel, *bufp); - } - else if (so->hashso_buc_populated && so->hashso_buc_split) - { - /* - * end of bucket, scan bucket being populated if there was a split in - * progress at the start of scan. - */ - *bufp = _hash_step_to_populated_bucket(scan); - *pagep = BufferGetPage(*bufp); - *opaquep = HashPageGetOpaque(*pagep); - } + return NULL; } /* * Cross over from the bucket being populated to the bucket being split, on - * whose primary page we have held a pin since _hash_first. Called on - * reaching the end of the populated bucket's chain while moving forward - * through it. + * whose primary page we have held a pin since _hash_first. * - * Returns the split bucket's primary page, pinned and share-locked, and - * sets hashso_buc_split to indicate that we are now scanning that bucket. + * Returns the split bucket's primary page, share-locked and pinned for the + * caller. Caller gets their own pin, not the scan's pin. */ static Buffer _hash_step_to_split_bucket(IndexScanDesc scan) @@ -283,34 +245,31 @@ _hash_step_to_split_bucket(IndexScanDesc scan) */ Assert(BufferIsValid(buf)); + /* + * Pin and share-lock the page for the traversal, which is what + * _hash_getbuf would do given its block number. We hold the buffer + * already, so just take another reference on it, next to the scan's own. + */ + IncrBufferRefCount(buf); LockBuffer(buf, BUFFER_LOCK_SHARE); PredicateLockPage(rel, BufferGetBlockNumber(buf), scan->xs_snapshot); - so->hashso_buc_split = true; - return buf; } /* - * Cross over from the bucket being split back to the bucket being - * populated, on whose primary page we have held a pin since _hash_first, - * and walk to the end of its chain (backward scans read chains tail to - * head). Called on reaching the start of the split bucket's chain while - * moving backward through it. + * Cross over from the bucket being split to the bucket being populated, on + * whose primary page we have held a pin since _hash_first, and walk to the + * end of its chain (backward scans read chains tail to head). * - * Returns the last page in the populated bucket's chain, pinned and - * share-locked, and clears hashso_buc_split: from here on the scan skips - * the moved-by-split tuples, whose originals it reads in the split - * bucket, and stops at the populated bucket's primary page instead of - * crossing again. + * Returns the last page in the populated bucket's chain, share-locked and + * pinned for the caller. Caller gets their own pin. */ static Buffer _hash_step_to_populated_bucket(IndexScanDesc scan) { HashScanOpaque so = (HashScanOpaque) scan->opaque; Buffer buf = so->hashso_bucket_buf; - Page page; - HashPageOpaque opaque; /* * buffer for bucket being populated must be valid as we acquire the pin @@ -318,36 +277,60 @@ _hash_step_to_populated_bucket(IndexScanDesc scan) */ Assert(BufferIsValid(buf)); + /* + * Pin and share-lock the page for the traversal, which is what + * _hash_getbuf would do given its block number. We hold the buffer + * already, so just take another reference on it, next to the scan's own. + */ + IncrBufferRefCount(buf); LockBuffer(buf, BUFFER_LOCK_SHARE); - page = BufferGetPage(buf); - opaque = HashPageGetOpaque(page); - /* move to the end of bucket chain */ + return _hash_chain_end(scan, buf); +} + +/* + * Walk from buf, a pinned and share-locked page of a bucket, to the last page + * of that bucket's chain, and return it pinned and share-locked. + */ +static Buffer +_hash_chain_end(IndexScanDesc scan, Buffer buf) +{ + Relation rel = scan->indexRelation; + HashPageOpaque opaque = HashPageGetOpaque(BufferGetPage(buf)); + while (BlockNumberIsValid(opaque->hasho_nextblkno)) - _hash_readnext(scan, &buf, &page, &opaque); + { + BlockNumber blkno = opaque->hasho_nextblkno; - so->hashso_buc_split = false; + _hash_relbuf(rel, buf); + + /* check for interrupts while we're not holding any buffer lock */ + CHECK_FOR_INTERRUPTS(); + + buf = _hash_getbuf(rel, blkno, HASH_READ, LH_OVERFLOW_PAGE); + opaque = HashPageGetOpaque(BufferGetPage(buf)); + } return buf; } /* - * _hash_first() -- Find the first item in a scan. + * _hash_first() -- Find the first batch of items in a scan. * - * We find the first item (or, if backward scan, the last item) in the - * index that satisfies the qualification associated with the scan - * descriptor. + * We find the first batch of items (or, if backward scan, the last + * batch) in the index that satisfies the qualification associated with + * the scan descriptor. * - * On successful exit, if the page containing current index tuple is an - * overflow page, both pin and lock are released whereas if it is a bucket - * page then it is pinned but not locked and data about the matching - * tuple(s) on the page has been loaded into so->currPos, - * scan->xs_heaptid is set to the heap TID of the current tuple. + * On successful exit, returns a batch containing matching items. + * Otherwise returns NULL, indicating that there are no further matches. + * No locks are ever held when we return. * - * On failure exit (no more tuples), we return false, with pin held on - * bucket page but no pins or locks held on overflow page. + * We keep our own pin on the primary bucket page, and on the primary + * page of the bucket being split when a split is in progress, until the + * scan is restarted or ended (except when we return NULL). A returned + * batch holds its own, separate pin on its page. */ -bool +IndexScanBatch _hash_first(IndexScanDesc scan, ScanDirection dir) { Relation rel = scan->indexRelation; @@ -358,7 +341,7 @@ _hash_first(IndexScanDesc scan, ScanDirection dir) Buffer buf; Page page; HashPageOpaque opaque; - HashScanPosItem *currItem; + IndexScanBatch firstbatch; pgstat_count_index_scan(rel); if (scan->instrument) @@ -388,7 +371,7 @@ _hash_first(IndexScanDesc scan, ScanDirection dir) * items in the index. */ if (cur->sk_flags & SK_ISNULL) - return false; + return NULL; /* * Okay to compute the hash key. We want to do this before acquiring any @@ -409,13 +392,29 @@ _hash_first(IndexScanDesc scan, ScanDirection dir) so->hashso_sk_hash = hashkey; + /* Allocate space for first batch before locking anything */ + firstbatch = batchscan_alloc(scan); + buf = _hash_getbucketbuf_from_hashkey(rel, hashkey, HASH_READ, NULL); PredicateLockPage(rel, BufferGetBlockNumber(buf), scan->xs_snapshot); page = BufferGetPage(buf); opaque = HashPageGetOpaque(page); bucket = opaque->hasho_bucket; + /* + * Keep our own pin on the primary bucket page until the scan is restarted + * or ended (see _hash_dropscanbuf), not just until we return the last + * batch. buf's original pin belongs to the page traversal, which + * releases it or passes it on to a batch. + * + * Holding the pin for the whole scan also gives index scans that use a + * scrollable cursor a consistent order. We make no guarantee about the + * order _across_ scans, though: a bucket split relocates tuples into the + * new bucket in a different order, and a squeeze repacks the chain (the + * bucket pin at least prevents that for the duration of a single scan). + */ so->hashso_bucket_buf = buf; + IncrBufferRefCount(buf); /* * If a bucket split is in progress, then while scanning the bucket being @@ -469,228 +468,67 @@ _hash_first(IndexScanDesc scan, ScanDirection dir) } } - /* If a backwards scan is requested, move to the end of the chain */ - if (ScanDirectionIsBackward(dir)) - { - /* - * Backward scans that start during split needs to start from end of - * bucket being split. - */ - while (BlockNumberIsValid(opaque->hasho_nextblkno) || - (so->hashso_buc_populated && !so->hashso_buc_split)) - _hash_readnext(scan, &buf, &page, &opaque); - } - - /* remember which buffer we have pinned, if any */ - Assert(BufferIsInvalid(so->currPos.buf)); - so->currPos.buf = buf; - - /* Now find all the tuples satisfying the qualification from a page */ - if (!_hash_readpage(scan, &buf, dir)) - return false; - - /* OK, itemIndex says what to return */ - currItem = &so->currPos.items[so->currPos.itemIndex]; - scan->xs_heaptid = currItem->heapTid; - - /* if we're here, _hash_readpage found a valid tuples */ - return true; + return _hash_readfirstpage(scan, firstbatch, buf, dir); } /* - * _hash_readpage() -- Load data from current index page into so->currPos + * _hash_readpage() -- Load data from an index page into batch * - * We scan all the items in the current index page and save them into - * so->currPos if it satisfies the qualification. If no matching items - * are found in the current page, we move to the next or previous page - * in a bucket chain as indicated by the direction. + * Caller must have pinned and share-locked buf; the buffer's state is not + * changed here. We save the items on the page that satisfy the + * qualification into batch, along with the page's neighbors, from which + * _hash_readnextpage continues the scan. bucSplit says whether the page is + * in the bucket being split rather than the one being populated, for a scan + * that started during a split. * - * Return true if any matching items are found else return false. + * Returns true if any matching items were found on the page, false if none. */ static bool -_hash_readpage(IndexScanDesc scan, Buffer *bufP, ScanDirection dir) +_hash_readpage(IndexScanDesc scan, Buffer buf, ScanDirection dir, + IndexScanBatch batch, bool bucSplit) { Relation rel = scan->indexRelation; HashScanOpaque so = (HashScanOpaque) scan->opaque; - Buffer buf; + HashBatchData *hashbatch = HashBatchGetData(scan, batch); Page page; HashPageOpaque opaque; - OffsetNumber offnum; - uint16 itemIndex; + OffsetNumber offnum, + maxoff; + IndexTuple itup; + int itemIndex; + bool skipmoved; - buf = *bufP; Assert(BufferIsValid(buf)); _hash_checkpage(rel, buf, LH_BUCKET_PAGE | LH_OVERFLOW_PAGE); page = BufferGetPage(buf); opaque = HashPageGetOpaque(page); - - so->currPos.buf = buf; - so->currPos.currPage = BufferGetBlockNumber(buf); - - if (ScanDirectionIsForward(dir)) - { - BlockNumber prev_blkno = InvalidBlockNumber; - - for (;;) - { - /* new page, locate starting position by binary search */ - offnum = _hash_binsearch(page, so->hashso_sk_hash); - - itemIndex = _hash_load_qualified_items(scan, page, offnum, dir); - - if (itemIndex != 0) - break; - - /* - * Could not find any matching tuples in the current page, move to - * the next page. Before leaving the current page, deal with any - * killed items. - */ - if (so->numKilled > 0) - _hash_kill_items(scan); - - /* - * If this is a primary bucket page, hasho_prevblkno is not a real - * block number. - */ - if (so->currPos.buf == so->hashso_bucket_buf || - so->currPos.buf == so->hashso_split_bucket_buf) - prev_blkno = InvalidBlockNumber; - else - prev_blkno = opaque->hasho_prevblkno; - - _hash_readnext(scan, &buf, &page, &opaque); - if (BufferIsValid(buf)) - { - so->currPos.buf = buf; - so->currPos.currPage = BufferGetBlockNumber(buf); - } - else - { - /* - * Remember next and previous block numbers for scrollable - * cursors to know the start position and return false - * indicating that no more matching tuples were found. Also, - * don't reset currPage or lsn, because we expect - * _hash_kill_items to be called for the old page after this - * function returns. - */ - so->currPos.prevPage = prev_blkno; - so->currPos.nextPage = InvalidBlockNumber; - so->currPos.buf = buf; - return false; - } - } - - so->currPos.firstItem = 0; - so->currPos.lastItem = itemIndex - 1; - so->currPos.itemIndex = 0; - } - else - { - BlockNumber next_blkno = InvalidBlockNumber; - - for (;;) - { - /* new page, locate starting position by binary search */ - offnum = _hash_binsearch_last(page, so->hashso_sk_hash); - - itemIndex = _hash_load_qualified_items(scan, page, offnum, dir); - - if (itemIndex != MaxIndexTuplesPerPage) - break; - - /* - * Could not find any matching tuples in the current page, move to - * the previous page. Before leaving the current page, deal with - * any killed items. - */ - if (so->numKilled > 0) - _hash_kill_items(scan); - - if (so->currPos.buf == so->hashso_bucket_buf || - so->currPos.buf == so->hashso_split_bucket_buf) - next_blkno = opaque->hasho_nextblkno; - - _hash_readprev(scan, &buf, &page, &opaque); - if (BufferIsValid(buf)) - { - so->currPos.buf = buf; - so->currPos.currPage = BufferGetBlockNumber(buf); - } - else - { - /* - * Remember next and previous block numbers for scrollable - * cursors to know the start position and return false - * indicating that no more matching tuples were found. Also, - * don't reset currPage or lsn, because we expect - * _hash_kill_items to be called for the old page after this - * function returns. - */ - so->currPos.prevPage = InvalidBlockNumber; - so->currPos.nextPage = next_blkno; - so->currPos.buf = buf; - return false; - } - } - - so->currPos.firstItem = itemIndex; - so->currPos.lastItem = MaxIndexTuplesPerPage - 1; - so->currPos.itemIndex = MaxIndexTuplesPerPage - 1; - } - - if (so->currPos.buf == so->hashso_bucket_buf || - so->currPos.buf == so->hashso_split_bucket_buf) - { - so->currPos.prevPage = InvalidBlockNumber; - so->currPos.nextPage = opaque->hasho_nextblkno; - LockBuffer(so->currPos.buf, BUFFER_LOCK_UNLOCK); - } - else - { - so->currPos.prevPage = opaque->hasho_prevblkno; - so->currPos.nextPage = opaque->hasho_nextblkno; - _hash_relbuf(rel, so->currPos.buf); - so->currPos.buf = InvalidBuffer; - } - - Assert(so->currPos.firstItem <= so->currPos.lastItem); - return true; -} - -/* - * Load all the qualified items from a current index page - * into so->currPos. Helper function for _hash_readpage. - */ -static int -_hash_load_qualified_items(IndexScanDesc scan, Page page, - OffsetNumber offnum, ScanDirection dir) -{ - HashScanOpaque so = (HashScanOpaque) scan->opaque; - IndexTuple itup; - int itemIndex; - OffsetNumber maxoff; - maxoff = PageGetMaxOffsetNumber(page); + hashbatch->buf = buf; + hashbatch->batchPage = BufferGetBlockNumber(buf); + hashbatch->bucSplit = bucSplit; + batch->dir = dir; + + /* + * A scan that started during a split skips the moved-by-split tuples in + * the bucket being populated, and reads their originals in the bucket + * being split instead. + */ + skipmoved = (so->hashso_buc_populated && !bucSplit); + + /* locate starting position by binary search, then load the items */ if (ScanDirectionIsForward(dir)) { /* load items[] in ascending order */ itemIndex = 0; + offnum = _hash_binsearch(page, so->hashso_sk_hash); while (offnum <= maxoff) { Assert(offnum >= FirstOffsetNumber); itup = (IndexTuple) PageGetItem(page, PageGetItemId(page, offnum)); - /* - * skip the tuples that are moved by split operation for the scan - * that has started when split was in progress. Also, skip the - * tuples that are marked as dead. - */ - if ((so->hashso_buc_populated && !so->hashso_buc_split && - (itup->t_info & INDEX_MOVED_BY_SPLIT_MASK)) || + if ((skipmoved && (itup->t_info & INDEX_MOVED_BY_SPLIT_MASK)) || (scan->ignore_killed_tuples && (ItemIdIsDead(PageGetItemId(page, offnum))))) { @@ -702,15 +540,12 @@ _hash_load_qualified_items(IndexScanDesc scan, Page page, _hash_checkqual(scan, itup)) { /* tuple is qualified, so remember it */ - _hash_saveitem(so, itemIndex, offnum, itup); + _hash_saveitem(batch, itemIndex, offnum, itup); itemIndex++; } else { - /* - * No more matching tuples exist in this page. so, exit while - * loop. - */ + /* No more matching tuples exist in this page */ break; } @@ -718,25 +553,22 @@ _hash_load_qualified_items(IndexScanDesc scan, Page page, } Assert(itemIndex <= MaxIndexTuplesPerPage); - return itemIndex; + batch->firstItem = 0; + batch->lastItem = itemIndex - 1; } else { /* load items[] in descending order */ itemIndex = MaxIndexTuplesPerPage; + offnum = _hash_binsearch_last(page, so->hashso_sk_hash); while (offnum >= FirstOffsetNumber) { Assert(offnum <= maxoff); itup = (IndexTuple) PageGetItem(page, PageGetItemId(page, offnum)); - /* - * skip the tuples that are moved by split operation for the scan - * that has started when split was in progress. Also, skip the - * tuples that are marked as dead. - */ - if ((so->hashso_buc_populated && !so->hashso_buc_split && - (itup->t_info & INDEX_MOVED_BY_SPLIT_MASK)) || + /* skip moved-by-split tuples and dead tuples */ + if ((skipmoved && (itup->t_info & INDEX_MOVED_BY_SPLIT_MASK)) || (scan->ignore_killed_tuples && (ItemIdIsDead(PageGetItemId(page, offnum))))) { @@ -749,14 +581,11 @@ _hash_load_qualified_items(IndexScanDesc scan, Page page, { itemIndex--; /* tuple is qualified, so remember it */ - _hash_saveitem(so, itemIndex, offnum, itup); + _hash_saveitem(batch, itemIndex, offnum, itup); } else { - /* - * No more matching tuples exist in this page. so, exit while - * loop. - */ + /* No more matching tuples exist in this page */ break; } @@ -764,17 +593,32 @@ _hash_load_qualified_items(IndexScanDesc scan, Page page, } Assert(itemIndex >= 0); - return itemIndex; + batch->firstItem = itemIndex; + batch->lastItem = MaxIndexTuplesPerPage - 1; } + + /* + * Remember the page's neighbors. A primary bucket page has none before + * it: its hasho_prevblkno holds a hashm_maxbucket value, not a block + * number (see HashPageOpaqueData). + */ + hashbatch->nextPage = opaque->hasho_nextblkno; + if (opaque->hasho_flag & LH_BUCKET_PAGE) + hashbatch->prevPage = InvalidBlockNumber; + else + hashbatch->prevPage = opaque->hasho_prevblkno; + + return (batch->firstItem <= batch->lastItem); } -/* Save an index item into so->currPos.items[itemIndex] */ +/* Save an index item into batch->items[itemIndex] */ static inline void -_hash_saveitem(HashScanOpaque so, int itemIndex, +_hash_saveitem(IndexScanBatch batch, int itemIndex, OffsetNumber offnum, IndexTuple itup) { - HashScanPosItem *currItem = &so->currPos.items[itemIndex]; + BatchMatchingItem *currItem = &batch->items[itemIndex]; - currItem->heapTid = itup->t_tid; + currItem->tableTid = itup->t_tid; currItem->indexOffset = offnum; + currItem->tupleOffset = 0; } diff --git a/src/backend/access/hash/hashutil.c b/src/backend/access/hash/hashutil.c index 1d1b05f87..331d5f4da 100644 --- a/src/backend/access/hash/hashutil.c +++ b/src/backend/access/hash/hashutil.c @@ -16,7 +16,6 @@ #include "access/hash.h" #include "access/reloptions.h" -#include "access/relscan.h" #include "port/pg_bitutils.h" #include "utils/lsyscache.h" #include "utils/rel.h" @@ -33,7 +32,7 @@ _hash_checkqual(IndexScanDesc scan, IndexTuple itup) /* * Currently, we can't check any of the scan conditions since we do not * have the original index entry value to supply to the sk_func. Always - * return true; we expect that hashgettuple already set the recheck flag + * return true; we expect that hashgetbatch already set the recheck flag * to make the main indexscan code do it. */ #ifdef NOT_USED @@ -505,128 +504,3 @@ _hash_get_newbucket_from_oldbucket(Relation rel, Bucket old_bucket, return new_bucket; } - -/* - * _hash_kill_items - set LP_DEAD state for items an indexscan caller has - * told us were killed. - * - * scan->opaque, referenced locally through so, contains information about the - * current page and killed tuples thereon (generally, this should only be - * called if so->numKilled > 0). - * - * The caller does not have a lock on the page and may or may not have the - * page pinned in a buffer. Note that read-lock is sufficient for setting - * LP_DEAD status (which is only a hint). - * - * The caller must have pin on bucket buffer, but may or may not have pin - * on overflow buffer, as indicated by HashScanPosIsPinned(so->currPos). - * - * We match items by heap TID before assuming they are the right ones to - * delete. - * - * There are never any scans active in a bucket at the time VACUUM begins, - * because VACUUM takes a cleanup lock on the primary bucket page and scans - * hold a pin. A scan can begin after VACUUM leaves the primary bucket page - * but before it finishes the entire bucket, but it can never pass VACUUM, - * because VACUUM always locks the next page before releasing the lock on - * the previous one. Therefore, we don't have to worry about accidentally - * killing a TID that has been reused for an unrelated tuple. - */ -void -_hash_kill_items(IndexScanDesc scan) -{ - HashScanOpaque so = (HashScanOpaque) scan->opaque; - Relation rel = scan->indexRelation; - BlockNumber blkno; - Buffer buf; - Page page; - HashPageOpaque opaque; - OffsetNumber offnum, - maxoff; - int numKilled = so->numKilled; - int i; - bool killedsomething = false; - bool havePin = false; - - Assert(so->numKilled > 0); - Assert(so->killedItems != NULL); - Assert(HashScanPosIsValid(so->currPos)); - - /* - * Always reset the scan state, so we don't look for same items on other - * pages. - */ - so->numKilled = 0; - - blkno = so->currPos.currPage; - if (HashScanPosIsPinned(so->currPos)) - { - /* - * We already have pin on this buffer, so, all we need to do is - * acquire lock on it. - */ - havePin = true; - buf = so->currPos.buf; - LockBuffer(buf, BUFFER_LOCK_SHARE); - } - else - buf = _hash_getbuf(rel, blkno, HASH_READ, LH_OVERFLOW_PAGE); - - page = BufferGetPage(buf); - opaque = HashPageGetOpaque(page); - maxoff = PageGetMaxOffsetNumber(page); - - for (i = 0; i < numKilled; i++) - { - int itemIndex = so->killedItems[i]; - HashScanPosItem *currItem = &so->currPos.items[itemIndex]; - - offnum = currItem->indexOffset; - - Assert(itemIndex >= so->currPos.firstItem && - itemIndex <= so->currPos.lastItem); - - while (offnum <= maxoff) - { - ItemId iid = PageGetItemId(page, offnum); - IndexTuple ituple = (IndexTuple) PageGetItem(page, iid); - - if (ItemPointerEquals(&ituple->t_tid, &currItem->heapTid)) - { - if (!killedsomething) - { - /* - * Use the hint bit infrastructure to check if we can - * update the page while just holding a share lock. If we - * are not allowed, there's no point continuing. - */ - if (!BufferBeginSetHintBits(buf)) - goto unlock_page; - } - - /* found the item */ - ItemIdMarkDead(iid); - killedsomething = true; - break; /* out of inner search loop */ - } - offnum = OffsetNumberNext(offnum); - } - } - - /* - * Since this can be redone later if needed, mark as dirty hint. Whenever - * we mark anything LP_DEAD, we also set the page's - * LH_PAGE_HAS_DEAD_TUPLES flag, which is likewise just a hint. - */ - if (killedsomething) - { - opaque->hasho_flag |= LH_PAGE_HAS_DEAD_TUPLES; - BufferFinishSetHintBits(buf, true, true); - } - -unlock_page: - if (havePin) - LockBuffer(so->currPos.buf, BUFFER_LOCK_UNLOCK); - else - _hash_relbuf(rel, buf); -} diff --git a/src/backend/access/index/batchscan.c b/src/backend/access/index/batchscan.c index d9c087311..7ac8ab373 100644 --- a/src/backend/access/index/batchscan.c +++ b/src/backend/access/index/batchscan.c @@ -137,7 +137,7 @@ batchscan_unlock(IndexScanDesc scan, IndexScanBatch batch, Buffer buf) { /* drop both the lock and the pin */ UnlockReleaseBuffer(buf); - batch->isGuarded = false; /* won't call amunguardbatch */ + Assert(!batch->isGuarded); /* won't call amunguardbatch */ } else { diff --git a/doc/src/sgml/indexam.sgml b/doc/src/sgml/indexam.sgml index 72224cd52..6dbb2f44b 100644 --- a/doc/src/sgml/indexam.sgml +++ b/doc/src/sgml/indexam.sgml @@ -735,7 +735,8 @@ ambeginscan (Relation indexRelation, scan->xs_recheck here, in ambeginscan, since the value applies to every item the scan returns. The value set here persists across any subsequent - amrescan calls. B-tree (always false) works this way. + amrescan calls. B-tree (always false) and hash (always + true) work this way. @@ -880,8 +881,9 @@ amgetbatch (IndexScanDesc scan, method. It is up to the table AM caller to decide when it should be released. Note also that amgetbatch functions must never modify the priorbatch parameter. The core - src/backend/access/nbtree/ implementation provides a - reference example of the amgetbatch interface. + src/backend/access/nbtree/ and + src/backend/access/hash/ implementations provide + reference examples of the amgetbatch interface. @@ -938,8 +940,8 @@ amunguardbatch (IndexScanDesc scan, is not even required to use the standard helper batchscan_unlock to manage it. In practice, though, most or all index AMs will use that helper and hold the simplest - possible interlock: each guarded B-tree batch keeps a single buffer pin - on the one index page the batch came from. See for details on buffer pin management during index scans. This function will be called at most once for each guarded batch; it is not called when the index AM has already unguarded the batch @@ -950,10 +952,11 @@ amunguardbatch (IndexScanDesc scan, The index AM may choose to retain its own buffer pins when this serves an - internal purpose (for example, maintaining a descent stack of pinned index - pages for reuse across amgetbatch calls). However, - any scheme that retains buffer pins managed by the index AM must be sure - to free the pins at an opportune point (at a minimum whenever + internal purpose (for example, the hash access method keeps a pin on the + scan's primary bucket page for the duration of the scan, which blocks the + concurrent bucket splits and compactions that would otherwise disrupt it). + However, any scheme that retains buffer pins managed by the index AM must + be sure to free the pins at an opportune point (at a minimum whenever amendscan is called, and typically when amrescan is called). It must also keep the number of retained pins fixed and small. @@ -982,8 +985,8 @@ amkillitemsbatch (IndexScanDesc scan, amgetbatch index AMs (those that don't can leave the field set to NULL), but doing so is recommended for performance, as it allows future scans to skip known-dead index entries. - The core index access method that currently supports - amgetbatch (B-tree) implements + Both core index access methods that currently support + amgetbatch (B-tree and hash) implement LP_DEAD marking, though third-party index access methods are free to choose whether to implement this feature. The table AM may call tableam_index_scanpos_killitem to mark dead items as @@ -1025,7 +1028,8 @@ amkillitemsbatch (IndexScanDesc scan, VACUUM recycling table TIDs — so it would be unsafe to assume that index entries still point to the same heap/table tuples. Since LP_DEAD marking is only an optimization - hint, it is always safe to skip it. B-tree uses this approach. + hint, it is always safe to skip it. Both B-tree and hash use this + approach. @@ -1115,8 +1119,8 @@ amgetbitmap (IndexScanDesc scan, amgetbitmap scans; during amgetbatch scans the priorbatch is strictly owned by the caller (the table AM), and the index AM must - never release it. See _bt_next for a reference - example. + never release it. See _bt_next and + _hash_next for reference examples. diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index b0f109712..8ec2daf64 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -1212,6 +1212,7 @@ Hash HashAggBatch HashAggSpill HashAllocFunc +HashBatchData HashBuildState HashBulkDeleteStreamPrivate HashCompareFunc @@ -1233,8 +1234,6 @@ HashPageStat HashPath HashScanOpaque HashScanOpaqueData -HashScanPosData -HashScanPosItem HashSkewBucket HashState HashValueFunc -- 2.55.0