From 0bec6002817a893051cf79656f3579da71e24d02 Mon Sep 17 00:00:00 2001 From: shihao zhong Date: Tue, 6 Oct 2026 20:56:14 -0400 Subject: [PATCH v10] Add pg_stat_tablespace statistics view pg_stat_io reports activity by object and context but not by tablespace, so there is no way to see which tablespace a workload is hitting. Add a stats kind that keeps, per tablespace, blocks read and hit, block read and write time, temporary file count and size, and the tuple counters. Block and tuple counts are added when relation and index stats are flushed. Block times are counted in pgstat_count_io_op_time_ext() and flushed with the pg_stat_io counters, so the checkpointer and the background writer report them too. This adds a pg_proc entry, so catversion needs a bump at commit. --- doc/src/sgml/monitoring.sgml | 229 +++++++++++++ src/backend/catalog/system_views.sql | 19 ++ src/backend/commands/tablespace.c | 7 + src/backend/storage/buffer/bufmgr.c | 20 +- src/backend/storage/buffer/localbuf.c | 10 +- src/backend/storage/file/fd.c | 8 +- src/backend/utils/activity/Makefile | 1 + src/backend/utils/activity/meson.build | 1 + src/backend/utils/activity/pgstat.c | 16 + src/backend/utils/activity/pgstat_database.c | 29 +- src/backend/utils/activity/pgstat_index.c | 8 + src/backend/utils/activity/pgstat_io.c | 23 ++ src/backend/utils/activity/pgstat_relation.c | 118 +++++++ src/backend/utils/activity/pgstat_shmem.c | 6 +- .../utils/activity/pgstat_tablespace.c | 321 ++++++++++++++++++ src/backend/utils/adt/pgstatfuncs.c | 79 ++++- src/backend/utils/cache/relcache.c | 16 + src/include/catalog/pg_proc.dat | 8 + src/include/pgstat.h | 35 +- src/include/utils/backend_status.h | 3 +- src/include/utils/pgstat_internal.h | 22 ++ src/include/utils/pgstat_kind.h | 15 +- .../expected/stats-tablespace-drop.out | 109 ++++++ src/test/isolation/isolation_schedule | 1 + .../specs/stats-tablespace-drop.spec | 52 +++ src/test/recovery/t/029_stats_restart.pl | 77 +++++ src/test/regress/expected/rules.out | 16 + src/test/regress/expected/stats.out | 154 ++++++++- src/test/regress/expected/tablespace.out | 54 +++ src/test/regress/sql/stats.sql | 80 +++++ src/test/regress/sql/tablespace.sql | 29 ++ src/tools/pgindent/typedefs.list | 3 + 32 files changed, 1532 insertions(+), 37 deletions(-) create mode 100644 src/backend/utils/activity/pgstat_tablespace.c create mode 100644 src/test/isolation/expected/stats-tablespace-drop.out create mode 100644 src/test/isolation/specs/stats-tablespace-drop.spec diff --git a/doc/src/sgml/monitoring.sgml b/doc/src/sgml/monitoring.sgml index 6337d2a3d25..f20e9640ca2 100644 --- a/doc/src/sgml/monitoring.sgml +++ b/doc/src/sgml/monitoring.sgml @@ -555,6 +555,15 @@ postgres 27093 0.0 0.0 30096 2752 ? Ss 11:34 0:00 postgres: ser + + pg_stat_tablespacepg_stat_tablespace + One row per tablespace, showing statistics about blocks, temporary + files and tuples for relations in that tablespace. See + + pg_stat_tablespace for details. + + + pg_stat_subscription_statspg_stat_subscription_stats One row per subscription, showing statistics about errors and conflicts. @@ -5732,6 +5741,220 @@ description | Waiting for a newly initialized WAL file to reach durable storage + + <structname>pg_stat_tablespace</structname> + + + pg_stat_tablespace + + + + The pg_stat_tablespace view will contain one row + for each tablespace in the cluster, showing statistics about blocks read + and hit, temporary file usage, and tuples read and written for relations + stored in that tablespace. + + + + Note that temporary files are attributed to the tablespace they were + created in, which is controlled by , + and is not necessarily the tablespace of any relation involved in the + query. + + + + <structname>pg_stat_tablespace</structname> View + + + + + + Column Type + + + Description + + + + + + + + + + + tablespace_id oid + + + OID of this tablespace + + + + + + + + tablespace_name name + + + Name of this tablespace + + + + + + + + blks_read bigint + + + Number of disk blocks read in this tablespace + + + + + + + + blks_hit bigint + + + Number of times disk blocks were found already in the buffer cache, so + that a read was not necessary (this only includes hits in the + PostgreSQL buffer cache, not the operating system's file system + cache) + + + + + + + + blk_read_time double precision + + + Time spent reading data file blocks in this tablespace, in + milliseconds (if is enabled, + otherwise zero) + + + + + + + + blk_write_time double precision + + + Time spent writing and extending data file blocks in this tablespace, + in milliseconds (if is enabled, + otherwise zero). Unlike + pg_stat_database.blk_write_time, + this includes writes done by the checkpointer and the background + writer, so the two do not add up. + + + + + + + + temp_files bigint + + + Number of temporary files created in this tablespace. All temporary + files are counted, regardless of why the temporary file was created, + and regardless of the setting. + + + + + + + + temp_bytes bigint + + + Total amount of data written to temporary files in this tablespace. + All temporary files are counted, regardless of why the temporary file + was created, and regardless of the + setting. + + + + + + + + tup_returned bigint + + + Number of live rows fetched by sequential scans and index entries + returned by index scans for relations in this tablespace + + + + + + + + tup_fetched bigint + + + Number of live rows fetched by index scans for relations in this + tablespace + + + + + + + + tup_inserted bigint + + + Number of rows inserted into relations in this tablespace + + + + + + + + tup_updated bigint + + + Number of rows updated in relations in this tablespace + + + + + + + + tup_deleted bigint + + + Number of rows deleted from relations in this tablespace + + + + + + + + stats_reset timestamp with time zone + + + Time at which these statistics were last reset + + + + + +
+
+ Statistics Functions @@ -6025,6 +6248,12 @@ description | Waiting for a newly initialized WAL file to reach durable storage pg_stat_slru view. + + + tablespace: Reset all the counters shown in the + pg_stat_tablespace view. + + wal: Reset all the counters shown in the diff --git a/src/backend/catalog/system_views.sql b/src/backend/catalog/system_views.sql index 809b9c0f1e4..ca6a3b9904f 100644 --- a/src/backend/catalog/system_views.sql +++ b/src/backend/catalog/system_views.sql @@ -1146,6 +1146,25 @@ CREATE VIEW pg_stat_replication_slots AS LATERAL pg_stat_get_replication_slot(slot_name) as s WHERE r.datoid IS NOT NULL; -- excluding physical slots +CREATE VIEW pg_stat_tablespace AS + SELECT + T.oid AS tablespace_id, + T.spcname AS tablespace_name, + S.blks_fetched - S.blks_hit AS blks_read, + S.blks_hit, + S.blk_read_time, + S.blk_write_time, + S.temp_files, + S.temp_bytes, + S.tup_returned, + S.tup_fetched, + S.tup_inserted, + S.tup_updated, + S.tup_deleted, + S.stats_reset + FROM pg_tablespace T, + LATERAL pg_stat_get_tablespace(T.oid) S; + CREATE VIEW pg_stat_database AS SELECT D.oid AS datid, diff --git a/src/backend/commands/tablespace.c b/src/backend/commands/tablespace.c index e01fb2db913..cda910e86f0 100644 --- a/src/backend/commands/tablespace.c +++ b/src/backend/commands/tablespace.c @@ -68,6 +68,7 @@ #include "commands/tablespace.h" #include "common/file_perm.h" #include "miscadmin.h" +#include "pgstat.h" #include "postmaster/bgwriter.h" #include "storage/fd.h" #include "storage/lmgr.h" @@ -363,6 +364,9 @@ CreateTableSpace(CreateTableSpaceStmt *stmt) /* Post creation hook for new tablespace */ InvokeObjectPostCreateHook(TableSpaceRelationId, tablespaceoid, 0); + /* Keep the cumulative stats system up-to-date */ + pgstat_create_tablespace(tablespaceoid); + create_tablespace_directories(location, tablespaceoid); /* Record the filesystem change in XLOG */ @@ -551,6 +555,9 @@ DropTableSpace(DropTableSpaceStmt *stmt) (void) XLogInsert(RM_TBLSPC_ID, XLOG_TBLSPC_DROP); } + /* Keep the cumulative stats system up-to-date */ + pgstat_drop_tablespace(tablespaceoid); + /* * Note: because we checked that the tablespace was empty, there should be * no need to worry about flushing shared buffers or free space map diff --git a/src/backend/storage/buffer/bufmgr.c b/src/backend/storage/buffer/bufmgr.c index 5c82865a084..b56f63e0a64 100644 --- a/src/backend/storage/buffer/bufmgr.c +++ b/src/backend/storage/buffer/bufmgr.c @@ -1833,8 +1833,9 @@ WaitReadBuffers(ReadBuffersOperation *operation) * itself was already counted earlier in AsyncReadBuffers() -- * either by us or by another backend if this is a foreign IO. */ - pgstat_count_io_op_time(io_object, io_context, IOOP_READ, - io_start, 0, 0); + pgstat_count_io_op_time_ext(io_object, io_context, IOOP_READ, + io_start, 0, 0, + operation->smgr->smgr_rlocator.locator.spcOid); } else { @@ -2153,8 +2154,9 @@ AsyncReadBuffers(ReadBuffersOperation *operation, int *nblocks_progress) smgrstartreadv(ioh, operation->smgr, forknum, blocknum, io_pages, io_buffers_len); - pgstat_count_io_op_time(io_object, io_context, IOOP_READ, - io_start, 1, io_buffers_len * BLCKSZ); + pgstat_count_io_op_time_ext(io_object, io_context, IOOP_READ, + io_start, 1, io_buffers_len * BLCKSZ, + operation->smgr->smgr_rlocator.locator.spcOid); if (persistence == RELPERSISTENCE_TEMP) pgBufferUsage.local_blks_read += io_buffers_len; @@ -3041,8 +3043,9 @@ ExtendBufferedRelShared(BufferManagerRelation bmr, if (!(flags & EB_SKIP_EXTENSION_LOCK)) UnlockRelationForExtension(bmr.rel, ExclusiveLock); - pgstat_count_io_op_time(IOOBJECT_RELATION, io_context, IOOP_EXTEND, - io_start, 1, extend_by * BLCKSZ); + pgstat_count_io_op_time_ext(IOOBJECT_RELATION, io_context, IOOP_EXTEND, + io_start, 1, extend_by * BLCKSZ, + BMR_GET_SMGR(bmr)->smgr_rlocator.locator.spcOid); /* Set BM_VALID, terminate IO, and wake up any waiters */ for (uint32 i = 0; i < extend_by; i++) @@ -4621,8 +4624,9 @@ FlushBuffer(BufferDesc *buf, SMgrRelation reln, IOObject io_object, * When a strategy is not in use, the write can only be a "regular" write * of a dirty shared buffer (IOCONTEXT_NORMAL IOOP_WRITE). */ - pgstat_count_io_op_time(io_object, io_context, - IOOP_WRITE, io_start, 1, BLCKSZ); + pgstat_count_io_op_time_ext(io_object, io_context, + IOOP_WRITE, io_start, 1, BLCKSZ, + reln->smgr_rlocator.locator.spcOid); pgBufferUsage.shared_blks_written++; diff --git a/src/backend/storage/buffer/localbuf.c b/src/backend/storage/buffer/localbuf.c index 4870c8e13d0..40420dd7827 100644 --- a/src/backend/storage/buffer/localbuf.c +++ b/src/backend/storage/buffer/localbuf.c @@ -212,8 +212,9 @@ FlushLocalBuffer(BufferDesc *bufHdr, SMgrRelation reln) false); /* Temporary table I/O does not use Buffer Access Strategies */ - pgstat_count_io_op_time(IOOBJECT_TEMP_RELATION, IOCONTEXT_NORMAL, - IOOP_WRITE, io_start, 1, BLCKSZ); + pgstat_count_io_op_time_ext(IOOBJECT_TEMP_RELATION, IOCONTEXT_NORMAL, + IOOP_WRITE, io_start, 1, BLCKSZ, + reln->smgr_rlocator.locator.spcOid); /* Mark not-dirty */ TerminateLocalBufferIO(bufHdr, true, 0, false); @@ -469,8 +470,9 @@ ExtendBufferedRelLocal(BufferManagerRelation bmr, /* actually extend relation */ smgrzeroextend(BMR_GET_SMGR(bmr), fork, first_block, extend_by, false); - pgstat_count_io_op_time(IOOBJECT_TEMP_RELATION, IOCONTEXT_NORMAL, IOOP_EXTEND, - io_start, 1, extend_by * BLCKSZ); + pgstat_count_io_op_time_ext(IOOBJECT_TEMP_RELATION, IOCONTEXT_NORMAL, IOOP_EXTEND, + io_start, 1, extend_by * BLCKSZ, + BMR_GET_SMGR(bmr)->smgr_rlocator.locator.spcOid); for (uint32 i = 0; i < extend_by; i++) { diff --git a/src/backend/storage/file/fd.c b/src/backend/storage/file/fd.c index 190c9974494..c1934005ad1 100644 --- a/src/backend/storage/file/fd.c +++ b/src/backend/storage/file/fd.c @@ -1515,7 +1515,7 @@ FileAccess(File file) static void ReportTemporaryFileUsage(const char *path, pgoff_t size) { - pgstat_report_tempfile(size); + pgstat_report_tempfile(size, path); if (log_temp_files >= 0) { @@ -1757,6 +1757,9 @@ OpenTemporaryFile(bool interXact) if (!interXact) RegisterTemporaryFile(file); + /* Its usage is reported when it is deleted, get ready for that */ + pgstat_prepare_report_tempfile(VfdCache[file].fileName); + return file; } @@ -1876,6 +1879,9 @@ PathNameCreateTemporaryFile(const char *path, bool error_on_failure) /* Register it for automatic close. */ RegisterTemporaryFile(file); + /* Its usage is reported when it is deleted, get ready for that */ + pgstat_prepare_report_tempfile(path); + return file; } diff --git a/src/backend/utils/activity/Makefile b/src/backend/utils/activity/Makefile index 2e32d1485d6..e3292a22219 100644 --- a/src/backend/utils/activity/Makefile +++ b/src/backend/utils/activity/Makefile @@ -34,6 +34,7 @@ OBJS = \ pgstat_shmem.o \ pgstat_slru.o \ pgstat_subscription.o \ + pgstat_tablespace.o \ pgstat_wal.o \ pgstat_xact.o \ wait_event.o \ diff --git a/src/backend/utils/activity/meson.build b/src/backend/utils/activity/meson.build index e6dcb2e26fc..3049bdf4887 100644 --- a/src/backend/utils/activity/meson.build +++ b/src/backend/utils/activity/meson.build @@ -19,6 +19,7 @@ backend_sources += files( 'pgstat_shmem.c', 'pgstat_slru.c', 'pgstat_subscription.c', + 'pgstat_tablespace.c', 'pgstat_wal.c', 'pgstat_xact.c', ) diff --git a/src/backend/utils/activity/pgstat.c b/src/backend/utils/activity/pgstat.c index 0b44a53ebe9..0706cf3af6d 100644 --- a/src/backend/utils/activity/pgstat.c +++ b/src/backend/utils/activity/pgstat.c @@ -301,6 +301,22 @@ static const PgStat_KindInfo pgstat_kind_builtin_infos[PGSTAT_KIND_BUILTIN_SIZE] .reset_timestamp_cb = pgstat_database_reset_timestamp_cb, }, + [PGSTAT_KIND_TABLESPACE] = { + .name = "tablespace", + + .fixed_amount = false, + .write_to_file = true, + /* so pg_stat_tablespace can be read from any database */ + .accessed_across_databases = true, + + .shared_size = sizeof(PgStatShared_Tablespace), + .shared_data_off = offsetof(PgStatShared_Tablespace, stats), + .shared_data_len = sizeof(((PgStatShared_Tablespace *) 0)->stats), + + .flush_static_cb = pgstat_flush_tablespace, + .reset_timestamp_cb = pgstat_tablespace_reset_timestamp_cb, + }, + [PGSTAT_KIND_RELATION] = { .name = "relation", diff --git a/src/backend/utils/activity/pgstat_database.c b/src/backend/utils/activity/pgstat_database.c index 7f3bc016593..0f4a3a8b7ca 100644 --- a/src/backend/utils/activity/pgstat_database.c +++ b/src/backend/utils/activity/pgstat_database.c @@ -214,13 +214,30 @@ pgstat_report_checksum_failures_in_db(Oid dboid, int failurecount) pgstat_unlock_entry(entry_ref); } +/* + * Prepare for reporting a temporary file that has just been created. + * + * The file is only reported when it is deleted. By the time those counts are + * flushed its tablespace can be gone, so the stats entry of the tablespace is + * created now, while the file keeps the tablespace from being dropped. + */ +void +pgstat_prepare_report_tempfile(const char *path) +{ + Oid spcoid = pgstat_tablespace_from_tempfile_path(path); + + if (OidIsValid(spcoid)) + pgstat_ensure_tablespace_entry(spcoid); +} + /* * Report creation of temporary file. */ void -pgstat_report_tempfile(size_t filesize) +pgstat_report_tempfile(size_t filesize, const char *path) { PgStat_StatDBEntry *dbent; + Oid spcoid; if (!pgstat_track_counts) return; @@ -228,6 +245,16 @@ pgstat_report_tempfile(size_t filesize) dbent = pgstat_prep_database_pending(MyDatabaseId); dbent->temp_bytes += filesize; dbent->temp_files++; + + spcoid = pgstat_tablespace_from_tempfile_path(path); + if (OidIsValid(spcoid)) + { + PgStat_StatTabspaceEntry *tsent; + + tsent = pgstat_prep_tablespace_pending(spcoid); + tsent->temp_bytes += filesize; + tsent->temp_files++; + } } /* diff --git a/src/backend/utils/activity/pgstat_index.c b/src/backend/utils/activity/pgstat_index.c index a1f9a4c6ac1..209f4800216 100644 --- a/src/backend/utils/activity/pgstat_index.c +++ b/src/backend/utils/activity/pgstat_index.c @@ -80,6 +80,12 @@ pgstat_index_flush_cb(PgStat_EntryRef *entry_ref, bool nowait) dbentry->blocks_fetched += lstats->idx.blocks_fetched; dbentry->blocks_hit += lstats->idx.blocks_hit; + /* + * Likewise for the tablespace the index lives in, which need not be the + * one holding the table. + */ + pgstat_relation_flush_tablespace(lstats); + return true; } @@ -93,6 +99,8 @@ pgstat_index_delete_pending_cb(PgStat_EntryRef *entry_ref) if (pending->relation) pgstat_unlink_relation(pending->relation); + if (pending->tsbase) + pfree(pending->tsbase); } /* diff --git a/src/backend/utils/activity/pgstat_io.c b/src/backend/utils/activity/pgstat_io.c index 8ec1aad5078..da8e6e18f69 100644 --- a/src/backend/utils/activity/pgstat_io.c +++ b/src/backend/utils/activity/pgstat_io.c @@ -113,6 +113,20 @@ pgstat_prepare_io_time(bool track_io_guc) void pgstat_count_io_op_time(IOObject io_object, IOContext io_context, IOOp io_op, instr_time start_time, uint32 cnt, uint64 bytes) +{ + pgstat_count_io_op_time_ext(io_object, io_context, io_op, start_time, + cnt, bytes, InvalidOid); +} + +/* + * Like pgstat_count_io_op_time() except the time is also credited to + * tablespace "spcoid", for pg_stat_tablespace. The buffer manager uses this + * for relation I/O; everything else passes InvalidOid. + */ +void +pgstat_count_io_op_time_ext(IOObject io_object, IOContext io_context, + IOOp io_op, instr_time start_time, + uint32 cnt, uint64 bytes, Oid spcoid) { if (!INSTR_TIME_IS_ZERO(start_time)) { @@ -147,6 +161,10 @@ pgstat_count_io_op_time(IOObject io_object, IOContext io_context, IOOp io_op, /* Add the per-backend count */ pgstat_count_backend_io_op_time(io_object, io_context, io_op, io_time); + + /* Add the per-tablespace count */ + if (OidIsValid(spcoid)) + pgstat_count_tablespace_io_op_time(spcoid, io_op, io_time); } pgstat_count_io_op(io_object, io_context, io_op, cnt, bytes); @@ -162,11 +180,16 @@ pgstat_fetch_stat_io(void) /* * Simpler wrapper of pgstat_io_flush_cb() + * + * This also flushes the per-tablespace I/O times counted along with the I/O + * statistics, as some of the processes calling this, like the checkpointer + * and the background writer, never call pgstat_report_stat(). */ void pgstat_flush_io(bool nowait) { (void) pgstat_io_flush_cb(nowait); + (void) pgstat_flush_tablespace(nowait); } /* diff --git a/src/backend/utils/activity/pgstat_relation.c b/src/backend/utils/activity/pgstat_relation.c index 5c70543ba88..9b048a745bc 100644 --- a/src/backend/utils/activity/pgstat_relation.c +++ b/src/backend/utils/activity/pgstat_relation.c @@ -37,6 +37,7 @@ typedef struct TwoPhasePgStatRecord PgStat_Counter updated_pre_truncdrop; PgStat_Counter deleted_pre_truncdrop; Oid id; /* table's OID */ + Oid tablespace_oid; /* table's tablespace OID */ bool shared; /* is it a shared catalog? */ bool truncdropped; /* was the relation truncated/dropped? */ } TwoPhasePgStatRecord; @@ -44,6 +45,8 @@ typedef struct TwoPhasePgStatRecord static PgStat_RelationStatus *pgstat_prep_relation_pending(PgStat_Kind kind, Oid rel_id, bool isshared); +static void pgstat_set_relation_tablespace(PgStat_RelationStatus *ps, + Oid spcoid); static void add_tabstat_xact_level(PgStat_RelationStatus *pgstat_info, int nest_level); static void ensure_tabstat_xact_level(PgStat_RelationStatus *pgstat_info); static void save_truncdrop_counters(PgStat_TableXactStatus *trans, bool is_drop); @@ -182,6 +185,14 @@ pgstat_assoc_relation(Relation rel) RelationGetRelid(rel), rel->rd_rel->relisshared); + /* + * Remember its tablespace. The relation is open, so the tablespace + * cannot be dropped right now, and its stats entry can be created. + */ + pgstat_set_relation_tablespace(rel->pgstat_info, rel->rd_locator.spcOid); + if (OidIsValid(rel->rd_locator.spcOid)) + pgstat_ensure_tablespace_entry(rel->rd_locator.spcOid); + /* don't allow link a stats to multiple relcache entries */ Assert(rel->pgstat_info->relation == NULL); @@ -206,6 +217,105 @@ pgstat_unlink_relation(Relation rel) rel->pgstat_info = NULL; } +/* + * Fill in the counts of a relation or index that are kept per tablespace. + * + * Rows changed by a transaction that is still open are not in here yet. They + * are added when the transaction ends, see AtEOXact_PgStat_Relations(). + */ +static void +pgstat_relation_tablespace_counts(const PgStat_RelationStatus *ps, + PgStat_StatTabspaceEntry *counts) +{ + memset(counts, 0, sizeof(*counts)); + + if (ps->kind == PGSTAT_KIND_INDEX) + { + counts->tuples_returned = ps->idx.tuples_returned; + counts->tuples_fetched = ps->idx.tuples_fetched; + counts->blocks_fetched = ps->idx.blocks_fetched; + counts->blocks_hit = ps->idx.blocks_hit; + } + else + { + counts->tuples_returned = ps->tab.counts.tuples_returned; + counts->tuples_fetched = ps->tab.counts.tuples_fetched; + counts->tuples_inserted = ps->tab.counts_xact.tuples_inserted; + counts->tuples_updated = ps->tab.counts_xact.tuples_updated; + counts->tuples_deleted = ps->tab.counts_xact.tuples_deleted; + counts->blocks_fetched = ps->tab.counts.blocks_fetched; + counts->blocks_hit = ps->tab.counts.blocks_hit; + } +} + +/* + * Add what a relation or index has counted to the pending statistics of the + * tablespace it lives in. + * + * If the relation was moved here from another tablespace, tsbase has what it + * had counted by then, which went to the old tablespace. + */ +void +pgstat_relation_flush_tablespace(PgStat_RelationStatus *ps) +{ + static const PgStat_StatTabspaceEntry zero = {0}; + const PgStat_StatTabspaceEntry *base = ps->tsbase ? ps->tsbase : &zero; + PgStat_StatTabspaceEntry cur; + PgStat_StatTabspaceEntry *tsentry; + + if (!OidIsValid(ps->tablespace_oid)) + return; + + pgstat_relation_tablespace_counts(ps, &cur); + + tsentry = pgstat_prep_tablespace_pending(ps->tablespace_oid); + tsentry->tuples_returned += cur.tuples_returned - base->tuples_returned; + tsentry->tuples_fetched += cur.tuples_fetched - base->tuples_fetched; + tsentry->tuples_inserted += cur.tuples_inserted - base->tuples_inserted; + tsentry->tuples_updated += cur.tuples_updated - base->tuples_updated; + tsentry->tuples_deleted += cur.tuples_deleted - base->tuples_deleted; + tsentry->blocks_fetched += cur.blocks_fetched - base->blocks_fetched; + tsentry->blocks_hit += cur.blocks_hit - base->blocks_hit; +} + +/* + * Record the tablespace a relation's pending statistics belong to. + * + * If that changes, what was counted so far is credited to the old tablespace. + */ +static void +pgstat_set_relation_tablespace(PgStat_RelationStatus *ps, Oid spcoid) +{ + if (ps->tablespace_oid == spcoid) + return; + + if (OidIsValid(ps->tablespace_oid)) + { + if (ps->tsbase == NULL) + ps->tsbase = MemoryContextAllocZero(GetMemoryChunkContext(ps), + sizeof(PgStat_StatTabspaceEntry)); + pgstat_relation_flush_tablespace(ps); + pgstat_relation_tablespace_counts(ps, ps->tsbase); + } + ps->tablespace_oid = spcoid; +} + +/* + * Refresh the tablespace recorded in a relation's pending statistics. + * + * pgstat_assoc_relation() records the tablespace when the pending entry is + * first created, but a relation can subsequently be moved to a different + * tablespace. RelationRebuildRelation() preserves pgstat_info across a rebuild, + * so without this the entry would keep crediting the old tablespace. + */ +void +pgstat_relation_update_tablespace(Relation rel) +{ + if (rel->pgstat_info != NULL) + pgstat_set_relation_tablespace(rel->pgstat_info, + rel->rd_locator.spcOid); +} + /* * Ensure that stats are dropped if transaction aborts. */ @@ -780,6 +890,7 @@ AtPrepare_PgStat_Relations(PgStat_SubXactStatus *xact_state) record.updated_pre_truncdrop = trans->updated_pre_truncdrop; record.deleted_pre_truncdrop = trans->deleted_pre_truncdrop; record.id = relstat->tab.id; + record.tablespace_oid = relstat->tablespace_oid; record.shared = relstat->tab.shared; record.truncdropped = trans->truncdropped; @@ -824,6 +935,7 @@ pgstat_twophase_postcommit(FullTransactionId fxid, uint16 info, /* Find or create a relstat entry for the rel */ pgstat_info = pgstat_prep_relation_pending(PGSTAT_KIND_RELATION, rec->id, rec->shared); + pgstat_set_relation_tablespace(pgstat_info, rec->tablespace_oid); /* Same math as in AtEOXact_PgStat, commit case */ pgstat_info->tab.counts_xact.tuples_inserted += rec->tuples_inserted; @@ -860,6 +972,7 @@ pgstat_twophase_postabort(FullTransactionId fxid, uint16 info, /* Find or create a relstat entry for the rel */ pgstat_info = pgstat_prep_relation_pending(PGSTAT_KIND_RELATION, rec->id, rec->shared); + pgstat_set_relation_tablespace(pgstat_info, rec->tablespace_oid); /* Same math as in AtEOXact_PgStat, abort case */ if (rec->truncdropped) @@ -975,6 +1088,9 @@ pgstat_relation_flush_cb(PgStat_EntryRef *entry_ref, bool nowait) dbentry->blocks_fetched += lstats->tab.counts.blocks_fetched; dbentry->blocks_hit += lstats->tab.counts.blocks_hit; + /* Likewise for the tablespace the relation lives in */ + pgstat_relation_flush_tablespace(lstats); + return true; } @@ -985,6 +1101,8 @@ pgstat_relation_delete_pending_cb(PgStat_EntryRef *entry_ref) if (pending->relation) pgstat_unlink_relation(pending->relation); + if (pending->tsbase) + pfree(pending->tsbase); } void diff --git a/src/backend/utils/activity/pgstat_shmem.c b/src/backend/utils/activity/pgstat_shmem.c index 1858d68f918..5013393acea 100644 --- a/src/backend/utils/activity/pgstat_shmem.c +++ b/src/backend/utils/activity/pgstat_shmem.c @@ -50,8 +50,6 @@ static void pgstat_drop_database_and_contents(Oid dboid); static void pgstat_free_entry(PgStatShared_HashEntry *shent, dshash_seq_status *hstat); static void pgstat_release_entry_ref(PgStat_HashKey key, PgStat_EntryRef *entry_ref, bool discard_pending); -static bool pgstat_need_entry_refs_gc(void); -static void pgstat_gc_entry_refs(void); static void pgstat_release_all_entry_refs(bool discard_pending); typedef bool (*ReleaseMatchCB) (PgStat_EntryRefHashEntry *, Datum data); static void pgstat_release_matching_entry_refs(bool discard_pending, ReleaseMatchCB match, Datum match_data); @@ -800,7 +798,7 @@ pgstat_request_entry_refs_gc(void) pg_atomic_fetch_add_u64(&pgStatLocal.shmem->gc_request_count, 1); } -static bool +bool pgstat_need_entry_refs_gc(void) { uint64 curage; @@ -816,7 +814,7 @@ pgstat_need_entry_refs_gc(void) return pgStatSharedRefAge != curage; } -static void +void pgstat_gc_entry_refs(void) { pgstat_entry_ref_hash_iterator i; diff --git a/src/backend/utils/activity/pgstat_tablespace.c b/src/backend/utils/activity/pgstat_tablespace.c new file mode 100644 index 00000000000..95c257b7315 --- /dev/null +++ b/src/backend/utils/activity/pgstat_tablespace.c @@ -0,0 +1,321 @@ +/* ------------------------------------------------------------------------- + * + * pgstat_tablespace.c + * Implementation of tablespace statistics. + * + * This file contains the implementation of tablespace statistics. It is kept + * separate from other statistics implementations for the sake of readability. + * + * Tablespace statistics are aggregated from several sources: block and tuple + * counts are folded in when relation and index statistics are flushed, block + * I/O times are counted along with the I/O statistics, and temporary file + * usage is reported by fd.c. + * + * All of them are accumulated in process-local memory first, and are added to + * the shared entries by pgstat_flush_tablespace(). The ordinary pending entry + * mechanism does not fit: pgstat_count_io_op_time_ext() also runs in the + * checkpointer and the background writer, which never call + * pgstat_report_stat(), and a pending entry created there would pin the + * shared entry for the life of the process. + * + * The flush never creates a shared entry. By then the tablespace may have + * been dropped, and an entry created for it would never be removed. Entries + * are created by pgstat_ensure_tablespace_entry() instead, at points where + * the tablespace cannot be dropped: in CREATE TABLESPACE, when a backend + * starts counting for a relation it has open, when a temporary file is + * created, and when a process counts its first block I/O time for the + * tablespace since its last flush. Counts for a tablespace without an entry + * are discarded, which is what happens to a tablespace dropped in the + * meantime. + * + * Copyright (c) 2026, PostgreSQL Global Development Group + * + * IDENTIFICATION + * src/backend/utils/activity/pgstat_tablespace.c + * ------------------------------------------------------------------------- + */ + +#include "postgres.h" + +#include "catalog/pg_tablespace_d.h" +#include "common/relpath.h" +#include "storage/fd.h" +#include "utils/memutils.h" +#include "utils/pgstat_internal.h" +#include "utils/timestamp.h" + + +/* + * Counts not yet flushed to shared memory, one element per tablespace. + * + * This is a plain array, searched linearly. A process works with few + * tablespaces between two flushes, and pgstat_count_tablespace_io_op_time() + * needs its element for every timed block I/O. I/O often alternates between + * tablespaces, for instance between a table and its index, or when the + * checkpointer balances its writes, so remembering only the tablespace used + * last in front of a hash table would not do. + */ +typedef struct PgStat_PendingTabspace +{ + Oid spcoid; + PgStat_StatTabspaceEntry counts; +} PgStat_PendingTabspace; + +static PgStat_PendingTabspace *pending_tabspaces = NULL; +static int npending_tabspaces = 0; +static int maxpending_tabspaces = 0; + + +/* + * Register the creation of a tablespace with the cumulative stats system. + * + * The entry is created right away, and is removed again if the transaction + * aborts. Any statistics left over from an earlier tablespace that happened + * to have the same OID are reset. + */ +void +pgstat_create_tablespace(Oid spcoid) +{ + pgstat_create_transactional(PGSTAT_KIND_TABLESPACE, InvalidOid, spcoid); + (void) pgstat_get_entry_ref(PGSTAT_KIND_TABLESPACE, InvalidOid, spcoid, + true, NULL); +} + +/* + * Remove entry for the tablespace being dropped. + */ +void +pgstat_drop_tablespace(Oid spcoid) +{ + pgstat_drop_transactional(PGSTAT_KIND_TABLESPACE, InvalidOid, spcoid); +} + +/* + * Make sure the shared entry for a tablespace exists. + * + * This must only be called while something keeps the tablespace from being + * dropped, such as a file in it that is in use. A call made after the drop + * would create an entry that nothing removes. + */ +void +pgstat_ensure_tablespace_entry(Oid spcoid) +{ + static Oid last_spcoid = InvalidOid; + + Assert(OidIsValid(spcoid)); + + /* + * This is called for every relation a backend starts counting for. An + * entry only goes away together with its tablespace, so there is no need + * to look again for the tablespace seen last. + */ + if (spcoid == last_spcoid) + return; + + (void) pgstat_get_entry_ref(PGSTAT_KIND_TABLESPACE, InvalidOid, spcoid, + true, NULL); + last_spcoid = spcoid; +} + +/* + * Fetch tablespace statistics. + */ +PgStat_StatTabspaceEntry * +pgstat_fetch_stat_tabspaceentry(Oid spcoid) +{ + return (PgStat_StatTabspaceEntry *) + pgstat_fetch_entry(PGSTAT_KIND_TABLESPACE, InvalidOid, spcoid, NULL); +} + +/* + * Find or create the process-local counts for a tablespace. + * + * If "ensure_entry" is set, the shared entry is created along with the local + * counts. See pgstat_ensure_tablespace_entry() for when that is allowed. + */ +static PgStat_PendingTabspace * +pgstat_get_pending_tablespace(Oid spcoid, bool ensure_entry) +{ + PgStat_PendingTabspace *pending; + + Assert(OidIsValid(spcoid)); + + for (int i = 0; i < npending_tabspaces; i++) + { + if (pending_tabspaces[i].spcoid == spcoid) + return &pending_tabspaces[i]; + } + + if (ensure_entry) + pgstat_ensure_tablespace_entry(spcoid); + + if (npending_tabspaces == maxpending_tabspaces) + { + int newmax = Max(8, maxpending_tabspaces * 2); + + if (pending_tabspaces == NULL) + pending_tabspaces = MemoryContextAlloc(TopMemoryContext, + newmax * sizeof(PgStat_PendingTabspace)); + else + pending_tabspaces = repalloc(pending_tabspaces, + newmax * sizeof(PgStat_PendingTabspace)); + maxpending_tabspaces = newmax; + } + + pending = &pending_tabspaces[npending_tabspaces++]; + pending->spcoid = spcoid; + memset(&pending->counts, 0, sizeof(pending->counts)); + + return pending; +} + +/* + * Prepare for reporting tablespace stats. + * + * The returned pointer is only good until the next call or flush. This does + * not create the shared entry, see pgstat_ensure_tablespace_entry(). + */ +PgStat_StatTabspaceEntry * +pgstat_prep_tablespace_pending(Oid spcoid) +{ + /* make sure pgstat_report_stat() gets to pgstat_flush_tablespace() */ + pgstat_report_fixed = true; + + return &pgstat_get_pending_tablespace(spcoid, false)->counts; +} + +/* + * Determine which tablespace a temporary file belongs to, based on its path. + * + * fd.c does not remember the tablespace a temporary file was created in -- by + * the time the file is deleted and its usage reported, only the path is still + * available. Rather than widen Vfd, we recover the OID from the path. + * + * TempTablespacePath() builds these paths, and produces just two shapes: one + * rooted at PG_TBLSPC_DIR for a real tablespace, and one rooted in the data + * directory for the default tablespace. Rather than hard-code the latter, we + * ask TempTablespacePath() itself what it looks like, so this stays correct if + * the layout ever changes. Note that it also maps the global tablespace onto + * the default one, so temporary files never belong to pg_global. + * + * Returns InvalidOid if the path is not recognized, in which case the caller + * simply does not attribute the file to any tablespace. + */ +Oid +pgstat_tablespace_from_tempfile_path(const char *path) +{ + char defaultpath[MAXPGPATH]; + + if (path == NULL) + return InvalidOid; + + if (strncmp(path, PG_TBLSPC_DIR_SLASH, strlen(PG_TBLSPC_DIR_SLASH)) == 0) + return atooid(path + strlen(PG_TBLSPC_DIR_SLASH)); + + TempTablespacePath(defaultpath, DEFAULTTABLESPACE_OID); + if (strncmp(path, defaultpath, strlen(defaultpath)) == 0) + return DEFAULTTABLESPACE_OID; + + return InvalidOid; +} + +/* + * Count the time of a block I/O against a tablespace. + * + * Called from pgstat_count_io_op_time_ext(), only when I/O timing is enabled. + * As in pg_stat_database, reads count as read time, and writes and extends + * count as write time. + * + * The caller has just done I/O on a relation in the tablespace, which thus + * cannot be dropped, so the shared entry can be created from here. This may + * be all that is ever counted for the tablespace, for instance in the + * checkpointer of a standby that runs no queries. + */ +void +pgstat_count_tablespace_io_op_time(Oid spcoid, IOOp io_op, instr_time io_time) +{ + PgStat_PendingTabspace *pending = pgstat_get_pending_tablespace(spcoid, true); + + if (io_op == IOOP_READ) + pending->counts.blk_read_time += INSTR_TIME_GET_MICROSEC(io_time); + else if (io_op == IOOP_WRITE || io_op == IOOP_EXTEND) + pending->counts.blk_write_time += INSTR_TIME_GET_MICROSEC(io_time); +} + +/* + * Flush out process-local counts. + * + * Returns true if some of them could not be flushed due to lock contention; + * those are kept and retried on the next call. + * + * Shared entries are not created here, see the file header. Counts for a + * tablespace that has no entry are dropped. + */ +bool +pgstat_flush_tablespace(bool nowait) +{ + int nkept = 0; + + /* + * Let go of references to entries dropped since the last call. The + * lookups in this file keep a reference to each entry they find, but + * dropped entries are only noticed in pgstat_get_entry_ref(). If the + * checkpointer still held such a reference at shutdown, the dropped entry + * would be there when the stats file is written. + */ + if (pgstat_need_entry_refs_gc()) + pgstat_gc_entry_refs(); + + for (int i = 0; i < npending_tabspaces; i++) + { + PgStat_PendingTabspace *pending = &pending_tabspaces[i]; + PgStat_EntryRef *entry_ref; + PgStat_StatTabspaceEntry *shent; + + entry_ref = pgstat_get_entry_ref(PGSTAT_KIND_TABLESPACE, InvalidOid, + pending->spcoid, false, NULL); + if (entry_ref == NULL) + continue; + + if (!pgstat_lock_entry(entry_ref, nowait)) + { + /* keep it for the next attempt */ + pending_tabspaces[nkept++] = *pending; + continue; + } + + shent = &((PgStatShared_Tablespace *) entry_ref->shared_stats)->stats; + +#define PGSTAT_ACCUM_TABSPACECOUNT(item) \ + (shent)->item += (pending->counts).item + + PGSTAT_ACCUM_TABSPACECOUNT(blocks_fetched); + PGSTAT_ACCUM_TABSPACECOUNT(blocks_hit); + PGSTAT_ACCUM_TABSPACECOUNT(blk_read_time); + PGSTAT_ACCUM_TABSPACECOUNT(blk_write_time); + PGSTAT_ACCUM_TABSPACECOUNT(temp_files); + PGSTAT_ACCUM_TABSPACECOUNT(temp_bytes); + PGSTAT_ACCUM_TABSPACECOUNT(tuples_returned); + PGSTAT_ACCUM_TABSPACECOUNT(tuples_fetched); + PGSTAT_ACCUM_TABSPACECOUNT(tuples_inserted); + PGSTAT_ACCUM_TABSPACECOUNT(tuples_updated); + PGSTAT_ACCUM_TABSPACECOUNT(tuples_deleted); + +#undef PGSTAT_ACCUM_TABSPACECOUNT + + pgstat_unlock_entry(entry_ref); + } + + npending_tabspaces = nkept; + + return nkept > 0; +} + +/* + * Reset stats reset timestamp. + */ +void +pgstat_tablespace_reset_timestamp_cb(PgStatShared_Common *header, TimestampTz ts) +{ + ((PgStatShared_Tablespace *) header)->stats.stat_reset_timestamp = ts; +} diff --git a/src/backend/utils/adt/pgstatfuncs.c b/src/backend/utils/adt/pgstatfuncs.c index 64b6f60516c..22b93980c27 100644 --- a/src/backend/utils/adt/pgstatfuncs.c +++ b/src/backend/utils/adt/pgstatfuncs.c @@ -2098,6 +2098,7 @@ pg_stat_reset_shared(PG_FUNCTION_ARGS) pgstat_reset_of_kind(PGSTAT_KIND_LOCK); XLogPrefetchResetStats(); pgstat_reset_of_kind(PGSTAT_KIND_SLRU); + pgstat_reset_of_kind(PGSTAT_KIND_TABLESPACE); pgstat_reset_of_kind(PGSTAT_KIND_WAL); PG_RETURN_VOID(); @@ -2119,13 +2120,15 @@ pg_stat_reset_shared(PG_FUNCTION_ARGS) XLogPrefetchResetStats(); else if (strcmp(target, "slru") == 0) pgstat_reset_of_kind(PGSTAT_KIND_SLRU); + else if (strcmp(target, "tablespace") == 0) + pgstat_reset_of_kind(PGSTAT_KIND_TABLESPACE); else if (strcmp(target, "wal") == 0) pgstat_reset_of_kind(PGSTAT_KIND_WAL); else ereport(ERROR, (errcode(ERRCODE_INVALID_PARAMETER_VALUE), errmsg("unrecognized reset target: \"%s\"", target), - errhint("Target must be \"archiver\", \"bgwriter\", \"checkpointer\", \"io\", \"lock\", \"recovery_prefetch\", \"slru\", or \"wal\"."))); + errhint("Target must be \"archiver\", \"bgwriter\", \"checkpointer\", \"io\", \"lock\", \"recovery_prefetch\", \"slru\", \"tablespace\", or \"wal\"."))); PG_RETURN_VOID(); } @@ -2493,6 +2496,80 @@ pg_stat_get_subscription_stats(PG_FUNCTION_ARGS) PG_RETURN_DATUM(HeapTupleGetDatum(heap_form_tuple(tupdesc, values, nulls))); } +/* + * Returns statistics for the given tablespace. If no statistics have been + * collected yet, all-zero stats are returned. + */ +Datum +pg_stat_get_tablespace(PG_FUNCTION_ARGS) +{ +#define PG_STAT_GET_TABLESPACE_COLS 12 + Oid spcoid = PG_GETARG_OID(0); + TupleDesc tupdesc; + Datum values[PG_STAT_GET_TABLESPACE_COLS] = {0}; + bool nulls[PG_STAT_GET_TABLESPACE_COLS] = {0}; + PgStat_StatTabspaceEntry *tsentry; + PgStat_StatTabspaceEntry allzero; + int i = 0; + + tupdesc = CreateTemplateTupleDesc(PG_STAT_GET_TABLESPACE_COLS); + TupleDescInitEntry(tupdesc, (AttrNumber) 1, "blks_fetched", + INT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 2, "blks_hit", + INT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 3, "blk_read_time", + FLOAT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 4, "blk_write_time", + FLOAT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 5, "temp_files", + INT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 6, "temp_bytes", + INT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 7, "tup_returned", + INT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 8, "tup_fetched", + INT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 9, "tup_inserted", + INT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 10, "tup_updated", + INT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 11, "tup_deleted", + INT8OID, -1, 0); + TupleDescInitEntry(tupdesc, (AttrNumber) 12, "stats_reset", + TIMESTAMPTZOID, -1, 0); + TupleDescFinalize(tupdesc); + tupdesc = BlessTupleDesc(tupdesc); + + tsentry = pgstat_fetch_stat_tabspaceentry(spcoid); + if (!tsentry) + { + memset(&allzero, 0, sizeof(PgStat_StatTabspaceEntry)); + tsentry = &allzero; + } + + values[i++] = Int64GetDatum(tsentry->blocks_fetched); + values[i++] = Int64GetDatum(tsentry->blocks_hit); + values[i++] = Float8GetDatum(pg_stat_us_to_ms(tsentry->blk_read_time)); + values[i++] = Float8GetDatum(pg_stat_us_to_ms(tsentry->blk_write_time)); + values[i++] = Int64GetDatum(tsentry->temp_files); + values[i++] = Int64GetDatum(tsentry->temp_bytes); + values[i++] = Int64GetDatum(tsentry->tuples_returned); + values[i++] = Int64GetDatum(tsentry->tuples_fetched); + values[i++] = Int64GetDatum(tsentry->tuples_inserted); + values[i++] = Int64GetDatum(tsentry->tuples_updated); + values[i++] = Int64GetDatum(tsentry->tuples_deleted); + + if (tsentry->stat_reset_timestamp == 0) + nulls[i] = true; + else + values[i] = TimestampTzGetDatum(tsentry->stat_reset_timestamp); + + Assert(i + 1 == PG_STAT_GET_TABLESPACE_COLS); + + PG_RETURN_DATUM(HeapTupleGetDatum(heap_form_tuple(tupdesc, values, nulls))); +#undef PG_STAT_GET_TABLESPACE_COLS +} + /* * Checks for presence of stats for object with provided kind, database oid, * object oid. diff --git a/src/backend/utils/cache/relcache.c b/src/backend/utils/cache/relcache.c index d8f04a05309..73bdf6aa098 100644 --- a/src/backend/utils/cache/relcache.c +++ b/src/backend/utils/cache/relcache.c @@ -1354,6 +1354,15 @@ RelationInitPhysicalAddr(Relation relation) else relation->rd_locator.dbOid = MyDatabaseId; + /* + * The relation may have been moved to another tablespace, so keep the + * tablespace recorded for statistics purposes in step. Paths that reload + * an entry in place, such as RelationReloadIndexInfo(), reach here with a + * live pgstat_info; RelationRebuildRelation() instead builds a fresh + * entry and swaps pgstat_info back afterwards, so it repeats this itself. + */ + pgstat_relation_update_tablespace(relation); + if (relation->rd_rel->relfilenode) { /* @@ -2753,6 +2762,13 @@ RelationRebuildRelation(Relation relation) /* pgstat_info / enabled must be preserved */ SWAPFIELD(struct PgStat_RelationStatus *, pgstat_info); SWAPFIELD(bool, pgstat_enabled); + + /* + * RelationInitPhysicalAddr() ran on the newly built entry, which had + * no pgstat_info yet, so redo the refresh now that the preserved + * pgstat_info and the rebuilt rd_locator are on the same entry. + */ + pgstat_relation_update_tablespace(relation); /* preserve old partition key if we have one */ if (keep_partkey) { diff --git a/src/include/catalog/pg_proc.dat b/src/include/catalog/pg_proc.dat index f46427258e3..98b725b95b6 100644 --- a/src/include/catalog/pg_proc.dat +++ b/src/include/catalog/pg_proc.dat @@ -6119,6 +6119,14 @@ proargnames => '{name,blks_zeroed,blks_hit,blks_read,blks_written,blks_exists,flushes,truncates,stats_reset}', prosrc => 'pg_stat_get_slru' }, +{ oid => '8463', descr => 'statistics: information about tablespace', + proname => 'pg_stat_get_tablespace', provolatile => 's', + proparallel => 'r', prorettype => 'record', proargtypes => 'oid', + proallargtypes => '{oid,int8,int8,float8,float8,int8,int8,int8,int8,int8,int8,int8,timestamptz}', + proargmodes => '{i,o,o,o,o,o,o,o,o,o,o,o,o}', + proargnames => '{spcoid,blks_fetched,blks_hit,blk_read_time,blk_write_time,temp_files,temp_bytes,tup_returned,tup_fetched,tup_inserted,tup_updated,tup_deleted,stats_reset}', + prosrc => 'pg_stat_get_tablespace' }, + { oid => '2978', descr => 'statistics: number of function calls', proname => 'pg_stat_get_function_calls', provolatile => 's', proparallel => 'r', prorettype => 'int8', proargtypes => 'oid', diff --git a/src/include/pgstat.h b/src/include/pgstat.h index 187d82c96fe..a7820ca6f04 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -224,7 +224,10 @@ typedef struct PgStat_IndexCounts typedef struct PgStat_RelationStatus { PgStat_Kind kind; /* PGSTAT_KIND_RELATION or PGSTAT_KIND_INDEX */ + Oid tablespace_oid; /* tablespace the relation lives in */ Relation relation; /* rel that is using this entry */ + /* counts credited to the tablespace it was moved out of, if any */ + struct PgStat_StatTabspaceEntry *tsbase; union { /* table counters */ @@ -275,7 +278,7 @@ typedef struct PgStat_TableXactStatus * ------------------------------------------------------------ */ -#define PGSTAT_FILE_FORMAT_ID 0x01A5BCBD +#define PGSTAT_FILE_FORMAT_ID 0x01A5BCBE typedef struct PgStat_ArchiverStats { @@ -459,6 +462,23 @@ typedef struct PgStat_StatDBEntry TimestampTz stat_reset_timestamp; } PgStat_StatDBEntry; +typedef struct PgStat_StatTabspaceEntry +{ + PgStat_Counter blocks_fetched; + PgStat_Counter blocks_hit; + PgStat_Counter blk_read_time; /* times in microseconds */ + PgStat_Counter blk_write_time; + PgStat_Counter temp_files; + PgStat_Counter temp_bytes; + PgStat_Counter tuples_returned; + PgStat_Counter tuples_fetched; + PgStat_Counter tuples_inserted; + PgStat_Counter tuples_updated; + PgStat_Counter tuples_deleted; + + TimestampTz stat_reset_timestamp; +} PgStat_StatTabspaceEntry; + typedef struct PgStat_StatFuncEntry { PgStat_Counter numcalls; @@ -703,6 +723,9 @@ extern instr_time pgstat_prepare_io_time(bool track_io_guc); extern void pgstat_count_io_op_time(IOObject io_object, IOContext io_context, IOOp io_op, instr_time start_time, uint32 cnt, uint64 bytes); +extern void pgstat_count_io_op_time_ext(IOObject io_object, IOContext io_context, + IOOp io_op, instr_time start_time, + uint32 cnt, uint64 bytes, Oid spcoid); extern PgStat_IO *pgstat_fetch_stat_io(void); extern const char *pgstat_get_io_context_name(IOContext io_context); @@ -779,6 +802,7 @@ extern void pgstat_copy_relation_stats(Relation dst, Relation src); extern void pgstat_init_relation(Relation rel); extern void pgstat_assoc_relation(Relation rel); extern void pgstat_unlink_relation(Relation rel); +extern void pgstat_relation_update_tablespace(Relation rel); extern void pgstat_report_vacuum(Relation rel, PgStat_Counter livetuples, PgStat_Counter deadtuples, @@ -885,6 +909,15 @@ extern PgStat_StatIdxEntry *pgstat_fetch_stat_idxentry_ext(bool shared, bool *may_free); +/* + * Functions in pgstat_tablespace.c + */ + +extern void pgstat_create_tablespace(Oid spcoid); +extern void pgstat_drop_tablespace(Oid spcoid); +extern PgStat_StatTabspaceEntry *pgstat_fetch_stat_tabspaceentry(Oid spcoid); + + /* * Functions in pgstat_replslot.c */ diff --git a/src/include/utils/backend_status.h b/src/include/utils/backend_status.h index a334e096e4a..b36749431c9 100644 --- a/src/include/utils/backend_status.h +++ b/src/include/utils/backend_status.h @@ -315,7 +315,8 @@ extern void pgstat_clear_backend_activity_snapshot(void); extern void pgstat_report_activity(BackendState state, const char *cmd_str); extern void pgstat_report_query_id(int64 query_id, bool force); extern void pgstat_report_plan_id(int64 plan_id, bool force); -extern void pgstat_report_tempfile(size_t filesize); +extern void pgstat_prepare_report_tempfile(const char *path); +extern void pgstat_report_tempfile(size_t filesize, const char *path); extern void pgstat_report_appname(const char *appname); extern void pgstat_report_xact_timestamp(TimestampTz tstamp); extern const char *pgstat_get_backend_current_activity(int pid, bool checkUser); diff --git a/src/include/utils/pgstat_internal.h b/src/include/utils/pgstat_internal.h index 201e57279b1..c7475913675 100644 --- a/src/include/utils/pgstat_internal.h +++ b/src/include/utils/pgstat_internal.h @@ -502,6 +502,12 @@ typedef struct PgStatShared_Database PgStat_StatDBEntry stats; } PgStatShared_Database; +typedef struct PgStatShared_Tablespace +{ + PgStatShared_Common header; + PgStat_StatTabspaceEntry stats; +} PgStatShared_Tablespace; + typedef struct PgStatShared_Relation { PgStatShared_Common header; @@ -753,6 +759,19 @@ extern bool pgstat_database_flush_cb(PgStat_EntryRef *entry_ref, bool nowait); extern void pgstat_database_reset_timestamp_cb(PgStatShared_Common *header, TimestampTz ts); +/* + * Functions in pgstat_tablespace.c + */ + +extern void pgstat_ensure_tablespace_entry(Oid spcoid); +extern PgStat_StatTabspaceEntry *pgstat_prep_tablespace_pending(Oid spcoid); +extern Oid pgstat_tablespace_from_tempfile_path(const char *path); +extern void pgstat_count_tablespace_io_op_time(Oid spcoid, IOOp io_op, + instr_time io_time); +extern bool pgstat_flush_tablespace(bool nowait); +extern void pgstat_tablespace_reset_timestamp_cb(PgStatShared_Common *header, TimestampTz ts); + + /* * Functions in pgstat_function.c */ @@ -789,6 +808,7 @@ extern void AtEOXact_PgStat_Relations(PgStat_SubXactStatus *xact_state, bool isC extern void AtEOSubXact_PgStat_Relations(PgStat_SubXactStatus *xact_state, bool isCommit, int nestDepth); extern void AtPrepare_PgStat_Relations(PgStat_SubXactStatus *xact_state); extern void PostPrepare_PgStat_Relations(PgStat_SubXactStatus *xact_state); +extern void pgstat_relation_flush_tablespace(PgStat_RelationStatus *ps); extern bool pgstat_relation_flush_cb(PgStat_EntryRef *entry_ref, bool nowait); extern void pgstat_relation_delete_pending_cb(PgStat_EntryRef *entry_ref); @@ -839,6 +859,8 @@ extern void pgstat_reset_matching_entries(bool (*do_reset) (PgStatShared_HashEnt TimestampTz ts); extern void pgstat_request_entry_refs_gc(void); +extern bool pgstat_need_entry_refs_gc(void); +extern void pgstat_gc_entry_refs(void); extern dsa_pointer pgstat_alloc_entry_body(PgStat_Kind kind); extern PgStatShared_Common *pgstat_init_entry(PgStat_Kind kind, PgStatShared_HashEntry *shhashent, diff --git a/src/include/utils/pgstat_kind.h b/src/include/utils/pgstat_kind.h index 45ca599d0dd..811b79f3b10 100644 --- a/src/include/utils/pgstat_kind.h +++ b/src/include/utils/pgstat_kind.h @@ -31,15 +31,16 @@ #define PGSTAT_KIND_REPLSLOT 5 /* per-slot statistics */ #define PGSTAT_KIND_SUBSCRIPTION 6 /* per-subscription statistics */ #define PGSTAT_KIND_BACKEND 7 /* per-backend statistics */ +#define PGSTAT_KIND_TABLESPACE 8 /* per-tablespace statistics */ /* stats for fixed-numbered objects */ -#define PGSTAT_KIND_ARCHIVER 8 -#define PGSTAT_KIND_BGWRITER 9 -#define PGSTAT_KIND_CHECKPOINTER 10 -#define PGSTAT_KIND_IO 11 -#define PGSTAT_KIND_LOCK 12 -#define PGSTAT_KIND_SLRU 13 -#define PGSTAT_KIND_WAL 14 +#define PGSTAT_KIND_ARCHIVER 9 +#define PGSTAT_KIND_BGWRITER 10 +#define PGSTAT_KIND_CHECKPOINTER 11 +#define PGSTAT_KIND_IO 12 +#define PGSTAT_KIND_LOCK 13 +#define PGSTAT_KIND_SLRU 14 +#define PGSTAT_KIND_WAL 15 #define PGSTAT_KIND_BUILTIN_MIN PGSTAT_KIND_DATABASE #define PGSTAT_KIND_BUILTIN_MAX PGSTAT_KIND_WAL diff --git a/src/test/isolation/expected/stats-tablespace-drop.out b/src/test/isolation/expected/stats-tablespace-drop.out new file mode 100644 index 00000000000..e6e8403c8f5 --- /dev/null +++ b/src/test/isolation/expected/stats-tablespace-drop.out @@ -0,0 +1,109 @@ +Parsed test spec with 2 sessions + +starting permutation: s2_check s1_read s2_drop_table s2_drop_ts s2_check s1_flush s2_check +pg_stat_force_next_flush +------------------------ + +(1 row) + +step s2_check: + SELECT EXISTS (SELECT FROM pg_tablespace t WHERE t.oid = o.oid) AS in_catalog, + pg_stat_have_stats('tablespace', 0, o.oid::int8) AS have_stats + FROM ts_oid o; + +in_catalog|have_stats +----------+---------- +t |t +(1 row) + +step s1_read: SELECT count(*) FROM t_drop; +count +----- + 100 +(1 row) + +step s2_drop_table: DROP TABLE t_drop; +step s2_drop_ts: DROP TABLESPACE regress_tblspace_drop; +step s2_check: + SELECT EXISTS (SELECT FROM pg_tablespace t WHERE t.oid = o.oid) AS in_catalog, + pg_stat_have_stats('tablespace', 0, o.oid::int8) AS have_stats + FROM ts_oid o; + +in_catalog|have_stats +----------+---------- +f |f +(1 row) + +step s1_flush: SELECT pg_stat_force_next_flush(); +pg_stat_force_next_flush +------------------------ + +(1 row) + +step s2_check: + SELECT EXISTS (SELECT FROM pg_tablespace t WHERE t.oid = o.oid) AS in_catalog, + pg_stat_have_stats('tablespace', 0, o.oid::int8) AS have_stats + FROM ts_oid o; + +in_catalog|have_stats +----------+---------- +f |f +(1 row) + +s2: NOTICE: table "t_drop" does not exist, skipping + +starting permutation: s2_check s1_temp s2_drop_table s2_drop_ts s2_check s1_flush s2_check +pg_stat_force_next_flush +------------------------ + +(1 row) + +step s2_check: + SELECT EXISTS (SELECT FROM pg_tablespace t WHERE t.oid = o.oid) AS in_catalog, + pg_stat_have_stats('tablespace', 0, o.oid::int8) AS have_stats + FROM ts_oid o; + +in_catalog|have_stats +----------+---------- +t |t +(1 row) + +step s1_temp: + SET temp_tablespaces = regress_tblspace_drop; + SET work_mem = '64kB'; + SELECT count(*) FROM (SELECT * FROM generate_series(1, 10000) g ORDER BY g DESC) s; + +count +----- +10000 +(1 row) + +step s2_drop_table: DROP TABLE t_drop; +step s2_drop_ts: DROP TABLESPACE regress_tblspace_drop; +step s2_check: + SELECT EXISTS (SELECT FROM pg_tablespace t WHERE t.oid = o.oid) AS in_catalog, + pg_stat_have_stats('tablespace', 0, o.oid::int8) AS have_stats + FROM ts_oid o; + +in_catalog|have_stats +----------+---------- +f |f +(1 row) + +step s1_flush: SELECT pg_stat_force_next_flush(); +pg_stat_force_next_flush +------------------------ + +(1 row) + +step s2_check: + SELECT EXISTS (SELECT FROM pg_tablespace t WHERE t.oid = o.oid) AS in_catalog, + pg_stat_have_stats('tablespace', 0, o.oid::int8) AS have_stats + FROM ts_oid o; + +in_catalog|have_stats +----------+---------- +f |f +(1 row) + +s2: NOTICE: table "t_drop" does not exist, skipping diff --git a/src/test/isolation/isolation_schedule b/src/test/isolation/isolation_schedule index bf9a2037c70..b31a8fd8553 100644 --- a/src/test/isolation/isolation_schedule +++ b/src/test/isolation/isolation_schedule @@ -108,6 +108,7 @@ test: vacuum-concurrent-drop test: vacuum-conflict test: vacuum-skip-locked test: stats +test: stats-tablespace-drop test: horizons test: predicate-hash test: predicate-gist diff --git a/src/test/isolation/specs/stats-tablespace-drop.spec b/src/test/isolation/specs/stats-tablespace-drop.spec new file mode 100644 index 00000000000..52d9cae017d --- /dev/null +++ b/src/test/isolation/specs/stats-tablespace-drop.spec @@ -0,0 +1,52 @@ +# Test that flushing pending statistics of a backend does not bring back the +# statistics entry of a tablespace that has been dropped in the meantime. + +setup { SET allow_in_place_tablespaces = on; } +setup { CREATE TABLESPACE regress_tblspace_drop LOCATION ''; } +setup +{ + CREATE TABLE t_drop (a int) TABLESPACE regress_tblspace_drop; + INSERT INTO t_drop SELECT generate_series(1, 100); + CREATE TABLE ts_oid AS + SELECT oid FROM pg_tablespace WHERE spcname = 'regress_tblspace_drop'; +} + +# Session teardowns run before this, so t_drop is already gone. +teardown { DROP TABLESPACE IF EXISTS regress_tblspace_drop; } + +session s1 +setup +{ + SET debug_parallel_query = off; + SELECT count(*) FROM t_drop; + SELECT pg_stat_force_next_flush(); +} +step s1_read { SELECT count(*) FROM t_drop; } +step s1_temp +{ + SET temp_tablespaces = regress_tblspace_drop; + SET work_mem = '64kB'; + SELECT count(*) FROM (SELECT * FROM generate_series(1, 10000) g ORDER BY g DESC) s; +} +step s1_flush { SELECT pg_stat_force_next_flush(); } + +session s2 +step s2_drop_table { DROP TABLE t_drop; } +step s2_drop_ts { DROP TABLESPACE regress_tblspace_drop; } +step s2_check +{ + SELECT EXISTS (SELECT FROM pg_tablespace t WHERE t.oid = o.oid) AS in_catalog, + pg_stat_have_stats('tablespace', 0, o.oid::int8) AS have_stats + FROM ts_oid o; +} +teardown +{ + DROP TABLE IF EXISTS t_drop; + DROP TABLE ts_oid; +} + +# s1 has counted a read when the tablespace goes away, and flushes afterwards +permutation s2_check s1_read s2_drop_table s2_drop_ts s2_check s1_flush s2_check + +# the same with a temporary file +permutation s2_check s1_temp s2_drop_table s2_drop_ts s2_check s1_flush s2_check diff --git a/src/test/recovery/t/029_stats_restart.pl b/src/test/recovery/t/029_stats_restart.pl index cdc427dbc78..2e408974274 100644 --- a/src/test/recovery/t/029_stats_restart.pl +++ b/src/test/recovery/t/029_stats_restart.pl @@ -293,6 +293,83 @@ cmp_ok( $wal_restart_immediate->{reset}, "$sect: reset timestamp is new"); + +## check that a tablespace written to by the checkpointer can be dropped +## without breaking the shutdown. The checkpointer keeps a reference to the +## stats entries it flushes block write times into, and writing out the stats +## file at shutdown expects dropped entries to be gone. + +$node->append_conf('postgresql.conf', + "track_io_timing = on\nallow_in_place_tablespaces = on"); +$node->restart; + +$sect = "dropped tablespace"; +$node->safe_psql($connect_db, + "CREATE TABLESPACE test_stats_tblspc LOCATION ''"); +my $spcoid = $node->safe_psql($connect_db, + "SELECT oid FROM pg_tablespace WHERE spcname = 'test_stats_tblspc'"); +$node->safe_psql($connect_db, + "CREATE TABLE tab_stats_tblspc TABLESPACE test_stats_tblspc AS SELECT generate_series(1,1000) AS a" +); +$node->safe_psql($connect_db, "CHECKPOINT"); +$node->safe_psql($connect_db, "DROP TABLE tab_stats_tblspc"); +$node->safe_psql($connect_db, "DROP TABLESPACE test_stats_tblspc"); + +my $log_offset = -s $node->logfile; +$node->stop; +ok( !$node->log_contains(qr/terminated by signal/, $log_offset), + "$sect: clean shutdown"); + +$node->start; +is(have_stats('tablespace', 0, $spcoid), + 'f', "$sect: tablespace stats do not exist"); + +## check that temporary files are counted for a tablespace without relations +## after a crash has discarded its stats entry. Flushing pending stats does +## not create the entry, so creating the temporary file has to. + +$sect = "tablespace after crash"; +$node->safe_psql($connect_db, + "CREATE TABLESPACE test_stats_tblspc_temp LOCATION ''"); +my $spcoid_temp = $node->safe_psql($connect_db, + "SELECT oid FROM pg_tablespace WHERE spcname = 'test_stats_tblspc_temp'"); + +$node->stop('immediate'); +$node->start; + +is(have_stats('tablespace', 0, $spcoid_temp), + 'f', "$sect: no stats for unused tablespace"); + +$node->safe_psql( + $connect_db, q[ + SET temp_tablespaces = test_stats_tblspc_temp; + SET work_mem = '64kB'; + SELECT count(*) FROM (SELECT * FROM generate_series(1, 10000) g ORDER BY g DESC) s; + SELECT pg_stat_force_next_flush();]); +is( $node->safe_psql( + $connect_db, + "SELECT temp_files > 0 FROM pg_stat_tablespace WHERE tablespace_name = 'test_stats_tblspc_temp'" + ), + 't', + "$sect: temporary files counted for tablespace without relations"); + +## check that block I/O times are counted for a tablespace after a crash, +## before any backend has opened a relation in it. Nothing else has created +## the stats entry at that point, so counting the I/O time has to. + +$sect = "tablespace I/O after crash"; +$node->safe_psql($connect_db, + "CREATE TABLE tab_stats_tblspc_io (a int) TABLESPACE test_stats_tblspc_temp" +); +$node->safe_psql($connect_db, + "INSERT INTO tab_stats_tblspc_io SELECT generate_series(1,1000)"); + +$node->stop('immediate'); +$node->start; + +is(have_stats('tablespace', 0, $spcoid_temp), + 't', "$sect: tablespace stats created by recovery"); + $node->stop; done_testing(); diff --git a/src/test/regress/expected/rules.out b/src/test/regress/expected/rules.out index 0addd043e68..23ad5c85db5 100644 --- a/src/test/regress/expected/rules.out +++ b/src/test/regress/expected/rules.out @@ -2375,6 +2375,22 @@ pg_stat_sys_tables| SELECT relid, stats_reset FROM pg_stat_all_tables WHERE ((schemaname = ANY (ARRAY['pg_catalog'::name, 'information_schema'::name])) OR (schemaname ~ '^pg_toast'::text)); +pg_stat_tablespace| SELECT t.oid AS tablespace_id, + t.spcname AS tablespace_name, + (s.blks_fetched - s.blks_hit) AS blks_read, + s.blks_hit, + s.blk_read_time, + s.blk_write_time, + s.temp_files, + s.temp_bytes, + s.tup_returned, + s.tup_fetched, + s.tup_inserted, + s.tup_updated, + s.tup_deleted, + s.stats_reset + FROM pg_tablespace t, + LATERAL pg_stat_get_tablespace(t.oid) s(blks_fetched, blks_hit, blk_read_time, blk_write_time, temp_files, temp_bytes, tup_returned, tup_fetched, tup_inserted, tup_updated, tup_deleted, stats_reset); pg_stat_user_functions| SELECT p.oid AS funcid, n.nspname AS schemaname, p.proname AS funcname, diff --git a/src/test/regress/expected/stats.out b/src/test/regress/expected/stats.out index 8b15471248b..efefb155c8b 100644 --- a/src/test/regress/expected/stats.out +++ b/src/test/regress/expected/stats.out @@ -121,14 +121,15 @@ SELECT id, name, fixed_amount, 5 | replslot | f | t | t 6 | subscription | f | t | t 7 | backend | f | t | f - 8 | archiver | t | f | t - 9 | bgwriter | t | f | t - 10 | checkpointer | t | f | t - 11 | io | t | f | t - 12 | lock | t | f | t - 13 | slru | t | f | t - 14 | wal | t | f | t -(14 rows) + 8 | tablespace | f | t | t + 9 | archiver | t | f | t + 10 | bgwriter | t | f | t + 11 | checkpointer | t | f | t + 12 | io | t | f | t + 13 | lock | t | f | t + 14 | slru | t | f | t + 15 | wal | t | f | t +(15 rows) -- ensure that both seqscan and indexscan plans are allowed SET enable_seqscan TO on; @@ -1266,7 +1267,7 @@ SELECT stats_reset > :'wal_reset_ts'::timestamptz FROM pg_stat_wal; -- Test error case for reset_shared with unknown stats type SELECT pg_stat_reset_shared('unknown'); ERROR: unrecognized reset target: "unknown" -HINT: Target must be "archiver", "bgwriter", "checkpointer", "io", "lock", "recovery_prefetch", "slru", or "wal". +HINT: Target must be "archiver", "bgwriter", "checkpointer", "io", "lock", "recovery_prefetch", "slru", "tablespace", or "wal". -- Test that reset works for pg_stat_database and pg_stat_database_conflicts -- Since pg_stat_database stats_reset starts out as NULL, reset it once first so that we -- have a baseline for comparison. The same for pg_stat_database_conflicts as it shares @@ -2154,4 +2155,139 @@ SELECT fastpath_exceeded > :backend_fastpath_exceeded_before (1 row) DROP TABLE part_test; +-- Test pg_stat_tablespace +-- pg_default and pg_global always exist +SELECT tablespace_name FROM pg_stat_tablespace + WHERE tablespace_name IN ('pg_default', 'pg_global') + ORDER BY tablespace_name; + tablespace_name +----------------- + pg_default + pg_global +(2 rows) + +-- Check only that the counters move, not by how much. pg_stat_tablespace +-- aggregates every relation in the tablespace and other sessions in this +-- parallel group are busy in pg_default too, so no exact value is +-- reproducible. I/O timings are only collected with track_io_timing on. +SET track_io_timing = on; +SELECT tup_inserted AS ts_ins_before, + tup_updated AS ts_upd_before, + tup_deleted AS ts_del_before, + tup_returned AS ts_ret_before, + blks_hit AS ts_hit_before, + blk_write_time AS ts_wtime_before + FROM pg_stat_tablespace WHERE tablespace_name = 'pg_default' \gset +-- Make the table big enough to be extended a number of times, as extending +-- a relation is what counts as write time here. +CREATE TABLE test_tablespace_stats (a int); +INSERT INTO test_tablespace_stats SELECT generate_series(1, 10000); +UPDATE test_tablespace_stats SET a = a + 1 WHERE a > 5000; +DELETE FROM test_tablespace_stats WHERE a > 9000; +SELECT count(*) > 0 FROM test_tablespace_stats; + ?column? +---------- + t +(1 row) + +SELECT pg_stat_force_next_flush(); + pg_stat_force_next_flush +-------------------------- + +(1 row) + +SELECT tup_inserted > :ts_ins_before AS inserts_counted, + tup_updated > :ts_upd_before AS updates_counted, + tup_deleted > :ts_del_before AS deletes_counted, + tup_returned > :ts_ret_before AS returns_counted, + blks_hit > :ts_hit_before AS hits_counted, + blk_write_time > :ts_wtime_before AS write_time_counted + FROM pg_stat_tablespace WHERE tablespace_name = 'pg_default'; + inserts_counted | updates_counted | deletes_counted | returns_counted | hits_counted | write_time_counted +-----------------+-----------------+-----------------+-----------------+--------------+-------------------- + t | t | t | t | t | t +(1 row) + +-- Block reads. Moving the table to another tablespace rewrites it without +-- going through shared buffers, so the SELECT has to read it back in, and +-- those reads belong to the new tablespace. Do this in a transaction to keep +-- autovacuum from reading the rewritten table first. +-- +-- blk_read_time is not checked. With asynchronous I/O, a read that completes +-- before the backend has to wait for it adds no measurable time. +SELECT blks_read AS ts_read_before + FROM pg_stat_tablespace WHERE tablespace_name = 'regress_tblspace' \gset +BEGIN; +ALTER TABLE test_tablespace_stats SET TABLESPACE regress_tblspace; +SELECT count(*) > 0 FROM test_tablespace_stats; + ?column? +---------- + t +(1 row) + +COMMIT; +SELECT pg_stat_force_next_flush(); + pg_stat_force_next_flush +-------------------------- + +(1 row) + +SELECT blks_read > :ts_read_before AS reads_counted + FROM pg_stat_tablespace WHERE tablespace_name = 'regress_tblspace'; + reads_counted +--------------- + t +(1 row) + +RESET track_io_timing; +DROP TABLE test_tablespace_stats; +-- Temporary files are attributed to the tablespace they were created in, +-- which without temp_tablespaces set is pg_default. +SELECT temp_files AS ts_tmpf_before, temp_bytes AS ts_tmpb_before + FROM pg_stat_tablespace WHERE tablespace_name = 'pg_default' \gset +SET work_mem = '64kB'; +SELECT count(*) > 0 FROM + (SELECT * FROM generate_series(1, 10000) AS s ORDER BY s DESC) AS foo; + ?column? +---------- + t +(1 row) + +RESET work_mem; +SELECT pg_stat_force_next_flush(); + pg_stat_force_next_flush +-------------------------- + +(1 row) + +SELECT temp_files > :ts_tmpf_before AS temp_files_counted, + temp_bytes > :ts_tmpb_before AS temp_bytes_counted + FROM pg_stat_tablespace WHERE tablespace_name = 'pg_default'; + temp_files_counted | temp_bytes_counted +--------------------+-------------------- + t | t +(1 row) + +-- Resetting sets the timestamp, and resetting again does not move it backwards +SELECT pg_stat_reset_shared('tablespace'); + pg_stat_reset_shared +---------------------- + +(1 row) + +SELECT stats_reset AS ts_reset_before FROM pg_stat_tablespace + WHERE tablespace_name = 'pg_default' \gset +SELECT pg_stat_reset_shared('tablespace'); + pg_stat_reset_shared +---------------------- + +(1 row) + +SELECT stats_reset >= :'ts_reset_before'::timestamptz FROM pg_stat_tablespace + WHERE tablespace_name = 'pg_default'; + ?column? +---------- + t +(1 row) + -- End of Stats Test diff --git a/src/test/regress/expected/tablespace.out b/src/test/regress/expected/tablespace.out index f0dd25cdf0c..28310ea8daa 100644 --- a/src/test/regress/expected/tablespace.out +++ b/src/test/regress/expected/tablespace.out @@ -951,6 +951,60 @@ ERROR: permission denied for tablespace regress_tblspace REINDEX (TABLESPACE regress_tblspace, CONCURRENTLY) TABLE tablespace_table; -- fail ERROR: permission denied for tablespace regress_tblspace RESET ROLE; +-- pg_stat_tablespace credits a relation moved to another tablespace to the +-- new one from the move on. What it did before belongs to the old one. +-- pgstat_info is kept across a relcache rebuild, so doing this in a +-- transaction that already used the relation exercises +-- pgstat_relation_update_tablespace(). The two tablespaces hold nothing but +-- this table, so the numbers are exact. +SET allow_in_place_tablespaces = true; +CREATE TABLESPACE regress_tblspace_stats1 LOCATION ''; +CREATE TABLESPACE regress_tblspace_stats2 LOCATION ''; +SELECT oid AS stats2_oid FROM pg_tablespace + WHERE spcname = 'regress_tblspace_stats2' \gset +CREATE TABLE tablespace_stats_move (a int) + WITH (autovacuum_enabled = off) TABLESPACE regress_tblspace_stats1; +INSERT INTO tablespace_stats_move SELECT generate_series(1, 10); +BEGIN; +SELECT count(*) FROM tablespace_stats_move; + count +------- + 10 +(1 row) + +ALTER TABLE tablespace_stats_move SET TABLESPACE regress_tblspace_stats2; +INSERT INTO tablespace_stats_move SELECT generate_series(1, 5); +SELECT count(*) FROM tablespace_stats_move; + count +------- + 15 +(1 row) + +COMMIT; +SELECT pg_stat_force_next_flush(); + pg_stat_force_next_flush +-------------------------- + +(1 row) + +SELECT tablespace_name, tup_inserted, tup_returned FROM pg_stat_tablespace + WHERE tablespace_name LIKE 'regress_tblspace_stats_' ORDER BY 1; + tablespace_name | tup_inserted | tup_returned +-------------------------+--------------+-------------- + regress_tblspace_stats1 | 10 | 10 + regress_tblspace_stats2 | 5 | 15 +(2 rows) + +DROP TABLE tablespace_stats_move; +DROP TABLESPACE regress_tblspace_stats1; +DROP TABLESPACE regress_tblspace_stats2; +SELECT pg_stat_have_stats('tablespace', 0, :stats2_oid); + pg_stat_have_stats +-------------------- + f +(1 row) + +RESET allow_in_place_tablespaces; ALTER TABLESPACE regress_tblspace RENAME TO regress_tblspace_renamed; ALTER TABLE ALL IN TABLESPACE regress_tblspace_renamed SET TABLESPACE pg_default; ALTER INDEX ALL IN TABLESPACE regress_tblspace_renamed SET TABLESPACE pg_default; diff --git a/src/test/regress/sql/stats.sql b/src/test/regress/sql/stats.sql index 674637e172b..65d0c91a687 100644 --- a/src/test/regress/sql/stats.sql +++ b/src/test/regress/sql/stats.sql @@ -1070,4 +1070,84 @@ SELECT fastpath_exceeded > :backend_fastpath_exceeded_before DROP TABLE part_test; +-- Test pg_stat_tablespace +-- pg_default and pg_global always exist +SELECT tablespace_name FROM pg_stat_tablespace + WHERE tablespace_name IN ('pg_default', 'pg_global') + ORDER BY tablespace_name; + +-- Check only that the counters move, not by how much. pg_stat_tablespace +-- aggregates every relation in the tablespace and other sessions in this +-- parallel group are busy in pg_default too, so no exact value is +-- reproducible. I/O timings are only collected with track_io_timing on. +SET track_io_timing = on; +SELECT tup_inserted AS ts_ins_before, + tup_updated AS ts_upd_before, + tup_deleted AS ts_del_before, + tup_returned AS ts_ret_before, + blks_hit AS ts_hit_before, + blk_write_time AS ts_wtime_before + FROM pg_stat_tablespace WHERE tablespace_name = 'pg_default' \gset + +-- Make the table big enough to be extended a number of times, as extending +-- a relation is what counts as write time here. +CREATE TABLE test_tablespace_stats (a int); +INSERT INTO test_tablespace_stats SELECT generate_series(1, 10000); +UPDATE test_tablespace_stats SET a = a + 1 WHERE a > 5000; +DELETE FROM test_tablespace_stats WHERE a > 9000; +SELECT count(*) > 0 FROM test_tablespace_stats; +SELECT pg_stat_force_next_flush(); + +SELECT tup_inserted > :ts_ins_before AS inserts_counted, + tup_updated > :ts_upd_before AS updates_counted, + tup_deleted > :ts_del_before AS deletes_counted, + tup_returned > :ts_ret_before AS returns_counted, + blks_hit > :ts_hit_before AS hits_counted, + blk_write_time > :ts_wtime_before AS write_time_counted + FROM pg_stat_tablespace WHERE tablespace_name = 'pg_default'; + +-- Block reads. Moving the table to another tablespace rewrites it without +-- going through shared buffers, so the SELECT has to read it back in, and +-- those reads belong to the new tablespace. Do this in a transaction to keep +-- autovacuum from reading the rewritten table first. +-- +-- blk_read_time is not checked. With asynchronous I/O, a read that completes +-- before the backend has to wait for it adds no measurable time. +SELECT blks_read AS ts_read_before + FROM pg_stat_tablespace WHERE tablespace_name = 'regress_tblspace' \gset +BEGIN; +ALTER TABLE test_tablespace_stats SET TABLESPACE regress_tblspace; +SELECT count(*) > 0 FROM test_tablespace_stats; +COMMIT; +SELECT pg_stat_force_next_flush(); + +SELECT blks_read > :ts_read_before AS reads_counted + FROM pg_stat_tablespace WHERE tablespace_name = 'regress_tblspace'; +RESET track_io_timing; + +DROP TABLE test_tablespace_stats; + +-- Temporary files are attributed to the tablespace they were created in, +-- which without temp_tablespaces set is pg_default. +SELECT temp_files AS ts_tmpf_before, temp_bytes AS ts_tmpb_before + FROM pg_stat_tablespace WHERE tablespace_name = 'pg_default' \gset + +SET work_mem = '64kB'; +SELECT count(*) > 0 FROM + (SELECT * FROM generate_series(1, 10000) AS s ORDER BY s DESC) AS foo; +RESET work_mem; +SELECT pg_stat_force_next_flush(); + +SELECT temp_files > :ts_tmpf_before AS temp_files_counted, + temp_bytes > :ts_tmpb_before AS temp_bytes_counted + FROM pg_stat_tablespace WHERE tablespace_name = 'pg_default'; + +-- Resetting sets the timestamp, and resetting again does not move it backwards +SELECT pg_stat_reset_shared('tablespace'); +SELECT stats_reset AS ts_reset_before FROM pg_stat_tablespace + WHERE tablespace_name = 'pg_default' \gset +SELECT pg_stat_reset_shared('tablespace'); +SELECT stats_reset >= :'ts_reset_before'::timestamptz FROM pg_stat_tablespace + WHERE tablespace_name = 'pg_default'; + -- End of Stats Test diff --git a/src/test/regress/sql/tablespace.sql b/src/test/regress/sql/tablespace.sql index c43a59e5957..5f56d2613f7 100644 --- a/src/test/regress/sql/tablespace.sql +++ b/src/test/regress/sql/tablespace.sql @@ -420,6 +420,35 @@ REINDEX (TABLESPACE regress_tblspace) TABLE tablespace_table; -- fail REINDEX (TABLESPACE regress_tblspace, CONCURRENTLY) TABLE tablespace_table; -- fail RESET ROLE; +-- pg_stat_tablespace credits a relation moved to another tablespace to the +-- new one from the move on. What it did before belongs to the old one. +-- pgstat_info is kept across a relcache rebuild, so doing this in a +-- transaction that already used the relation exercises +-- pgstat_relation_update_tablespace(). The two tablespaces hold nothing but +-- this table, so the numbers are exact. +SET allow_in_place_tablespaces = true; +CREATE TABLESPACE regress_tblspace_stats1 LOCATION ''; +CREATE TABLESPACE regress_tblspace_stats2 LOCATION ''; +SELECT oid AS stats2_oid FROM pg_tablespace + WHERE spcname = 'regress_tblspace_stats2' \gset +CREATE TABLE tablespace_stats_move (a int) + WITH (autovacuum_enabled = off) TABLESPACE regress_tblspace_stats1; +INSERT INTO tablespace_stats_move SELECT generate_series(1, 10); +BEGIN; +SELECT count(*) FROM tablespace_stats_move; +ALTER TABLE tablespace_stats_move SET TABLESPACE regress_tblspace_stats2; +INSERT INTO tablespace_stats_move SELECT generate_series(1, 5); +SELECT count(*) FROM tablespace_stats_move; +COMMIT; +SELECT pg_stat_force_next_flush(); +SELECT tablespace_name, tup_inserted, tup_returned FROM pg_stat_tablespace + WHERE tablespace_name LIKE 'regress_tblspace_stats_' ORDER BY 1; +DROP TABLE tablespace_stats_move; +DROP TABLESPACE regress_tblspace_stats1; +DROP TABLESPACE regress_tblspace_stats2; +SELECT pg_stat_have_stats('tablespace', 0, :stats2_oid); +RESET allow_in_place_tablespaces; + ALTER TABLESPACE regress_tblspace RENAME TO regress_tblspace_renamed; ALTER TABLE ALL IN TABLESPACE regress_tblspace_renamed SET TABLESPACE pg_default; diff --git a/src/tools/pgindent/typedefs.list b/src/tools/pgindent/typedefs.list index d001f56efae..34c71a28b17 100644 --- a/src/tools/pgindent/typedefs.list +++ b/src/tools/pgindent/typedefs.list @@ -2305,6 +2305,7 @@ PgStatShared_Relation PgStatShared_ReplSlot PgStatShared_SLRU PgStatShared_Subscription +PgStatShared_Tablespace PgStatShared_Wal PgStat_ArchiverStats PgStat_Backend @@ -2329,6 +2330,7 @@ PgStat_LockEntry PgStat_PendingDroppedStatsItem PgStat_PendingIO PgStat_PendingLock +PgStat_PendingTabspace PgStat_RelationStatus PgStat_SLRUStats PgStat_ShmemControl @@ -2342,6 +2344,7 @@ PgStat_StatIdxEntry PgStat_StatReplSlotEntry PgStat_StatSubEntry PgStat_StatTabEntry +PgStat_StatTabspaceEntry PgStat_StatsFileOp PgStat_SubXactStatus PgStat_TableCounts -- 2.37.1 (Apple Git-137.1)