From d806686e71808878e8b02b14dd5dbc4d35802a8b Mon Sep 17 00:00:00 2001 From: Nathan Bossart Date: Tue, 11 Feb 2025 16:38:14 -0600 Subject: [PATCH 01/18] Add is_analyze parameter to vacuum_delay_point(). This function is used in both vacuum and analyze code paths, and a follow-up commit will require distinguishing between the two. This commit forces callers to specify whether they are in a vacuum or analyze path, but it does not use that information for anything yet. Author: Nathan Bossart Co-authored-by: Bertrand Drouvot Discussion: https://postgr.es/m/ZmaXmWDL829fzAVX%40ip-10-97-1-34.eu-west-3.compute.internal Backported from PostgreSQL 18 (commit e5b0b0ce150972bf162a059430d84e5f8e07cf30) as a prerequisite of the vacuum statistics series. Cloudberry adaptation: the append-optimized and AOCS sample-row acquisition and compaction call vacuum_delay_point() too; the former pass is_analyze = true, the latter false. (cherry picked from commit e07fe9839113a7264d09c5b51c093c611f2f77cc) (cherry picked from commit ffe16da6e4b80096316ec815ca7f219d2c6473a2) --- contrib/bloom/blvacuum.c | 4 ++-- contrib/file_fdw/file_fdw.c | 2 +- src/backend/access/aocs/aocs_compaction.c | 2 +- src/backend/access/aocs/aocsam_handler.c | 2 +- src/backend/access/appendonly/appendonly_compaction.c | 2 +- src/backend/access/appendonly/appendonlyam_handler.c | 2 +- src/backend/access/gin/ginfast.c | 6 +++--- src/backend/access/gin/ginvacuum.c | 6 +++--- src/backend/access/gist/gistvacuum.c | 2 +- src/backend/access/hash/hash.c | 2 +- src/backend/access/heap/vacuumlazy.c | 8 ++++---- src/backend/access/nbtree/nbtree.c | 2 +- src/backend/access/spgist/spgvacuum.c | 4 ++-- src/backend/commands/analyze.c | 10 +++++----- src/backend/commands/vacuum.c | 2 +- src/backend/tsearch/ts_typanalyze.c | 2 +- src/backend/utils/adt/array_typanalyze.c | 2 +- src/backend/utils/adt/rangetypes_typanalyze.c | 2 +- src/include/commands/vacuum.h | 2 +- 19 files changed, 32 insertions(+), 32 deletions(-) diff --git a/contrib/bloom/blvacuum.c b/contrib/bloom/blvacuum.c index 88b0a6d2900..d8873f96822 100644 --- a/contrib/bloom/blvacuum.c +++ b/contrib/bloom/blvacuum.c @@ -61,7 +61,7 @@ blbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats, *itupPtr, *itupEnd; - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, info->strategy); @@ -191,7 +191,7 @@ blvacuumcleanup(IndexVacuumInfo *info, IndexBulkDeleteResult *stats) Buffer buffer; Page page; - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, info->strategy); diff --git a/contrib/file_fdw/file_fdw.c b/contrib/file_fdw/file_fdw.c index cce94a5b335..5e870115661 100644 --- a/contrib/file_fdw/file_fdw.c +++ b/contrib/file_fdw/file_fdw.c @@ -1170,7 +1170,7 @@ file_acquire_sample_rows(Relation onerel, int elevel, for (;;) { /* Check for user-requested abort or sleep */ - vacuum_delay_point(); + vacuum_delay_point(true); /* Fetch next row */ MemoryContextReset(tupcontext); diff --git a/src/backend/access/aocs/aocs_compaction.c b/src/backend/access/aocs/aocs_compaction.c index e6ecea73de1..41919408c01 100644 --- a/src/backend/access/aocs/aocs_compaction.c +++ b/src/backend/access/aocs/aocs_compaction.c @@ -323,7 +323,7 @@ AOCSSegmentFileFullCompaction(Relation aorel, tupleCount++; if (VacuumCostActive && tupleCount % tuplePerPage == 0) { - vacuum_delay_point(); + vacuum_delay_point(false); } /* diff --git a/src/backend/access/aocs/aocsam_handler.c b/src/backend/access/aocs/aocsam_handler.c index 4b3cd2a52ef..83bba9e0443 100644 --- a/src/backend/access/aocs/aocsam_handler.c +++ b/src/backend/access/aocs/aocsam_handler.c @@ -1681,7 +1681,7 @@ aoco_acquire_sample_rows(Relation onerel, int elevel, HeapTuple *rows, { aocoscan->targrow = RowSampler_Next(&rs); - vacuum_delay_point(); + vacuum_delay_point(true); if (aocs_get_target_tuple(aocoscan, aocoscan->targrow, slot)) { diff --git a/src/backend/access/appendonly/appendonly_compaction.c b/src/backend/access/appendonly/appendonly_compaction.c index 05b2143b246..5ad525883b4 100644 --- a/src/backend/access/appendonly/appendonly_compaction.c +++ b/src/backend/access/appendonly/appendonly_compaction.c @@ -506,7 +506,7 @@ AppendOnlySegmentFileFullCompaction(Relation aorel, tupleCount++; if (VacuumCostActive && tupleCount % tuplePerPage == 0) { - vacuum_delay_point(); + vacuum_delay_point(false); } } diff --git a/src/backend/access/appendonly/appendonlyam_handler.c b/src/backend/access/appendonly/appendonlyam_handler.c index 715cebd7579..ad299fecb51 100644 --- a/src/backend/access/appendonly/appendonlyam_handler.c +++ b/src/backend/access/appendonly/appendonlyam_handler.c @@ -1563,7 +1563,7 @@ appendonly_acquire_sample_rows(Relation onerel, int elevel, HeapTuple *rows, { aoscan->targrow = RowSampler_Next(&rs); - vacuum_delay_point(); + vacuum_delay_point(true); if (appendonly_get_target_tuple(aoscan, aoscan->targrow, slot)) { diff --git a/src/backend/access/gin/ginfast.c b/src/backend/access/gin/ginfast.c index 3f84e90b260..317041a454f 100644 --- a/src/backend/access/gin/ginfast.c +++ b/src/backend/access/gin/ginfast.c @@ -894,7 +894,7 @@ ginInsertCleanup(GinState *ginstate, bool full_clean, */ processPendingPage(&accum, &datums, page, FirstOffsetNumber); - vacuum_delay_point(); + vacuum_delay_point(false); /* * Is it time to flush memory to disk? Flush if we are at the end of @@ -931,7 +931,7 @@ ginInsertCleanup(GinState *ginstate, bool full_clean, { ginEntryInsert(ginstate, attnum, key, category, list, nlist, NULL); - vacuum_delay_point(); + vacuum_delay_point(false); } /* @@ -1004,7 +1004,7 @@ ginInsertCleanup(GinState *ginstate, bool full_clean, /* * Read next page in pending list */ - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBuffer(index, blkno); LockBuffer(buffer, GIN_SHARE); page = BufferGetPage(buffer); diff --git a/src/backend/access/gin/ginvacuum.c b/src/backend/access/gin/ginvacuum.c index a276eb020b5..9d0d5218392 100644 --- a/src/backend/access/gin/ginvacuum.c +++ b/src/backend/access/gin/ginvacuum.c @@ -663,12 +663,12 @@ ginbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats, UnlockReleaseBuffer(buffer); } - vacuum_delay_point(); + vacuum_delay_point(false); for (i = 0; i < nRoot; i++) { ginVacuumPostingTree(&gvs, rootOfPostingTree[i]); - vacuum_delay_point(); + vacuum_delay_point(false); } if (blkno == InvalidBlockNumber) /* rightmost page */ @@ -749,7 +749,7 @@ ginvacuumcleanup(IndexVacuumInfo *info, IndexBulkDeleteResult *stats) Buffer buffer; Page page; - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, info->strategy); diff --git a/src/backend/access/gist/gistvacuum.c b/src/backend/access/gist/gistvacuum.c index 0663193531a..8a9805207a8 100644 --- a/src/backend/access/gist/gistvacuum.c +++ b/src/backend/access/gist/gistvacuum.c @@ -277,7 +277,7 @@ gistvacuumpage(GistVacState *vstate, BlockNumber blkno, BlockNumber orig_blkno) recurse_to = InvalidBlockNumber; /* call vacuum_delay_point while not holding any buffer lock */ - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(rel, MAIN_FORKNUM, blkno, RBM_NORMAL, info->strategy); diff --git a/src/backend/access/hash/hash.c b/src/backend/access/hash/hash.c index 9ab8b829a1e..470daa022dc 100644 --- a/src/backend/access/hash/hash.c +++ b/src/backend/access/hash/hash.c @@ -730,7 +730,7 @@ hashbucketcleanup(Relation rel, Bucket cur_bucket, Buffer bucket_buf, bool retain_pin = false; bool clear_dead_marking = false; - vacuum_delay_point(); + vacuum_delay_point(false); page = BufferGetPage(buf); opaque = (HashPageOpaque) PageGetSpecialPointer(page); diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 44538b626c6..76758799b22 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -1075,7 +1075,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) if ((vmstatus & VISIBILITYMAP_ALL_VISIBLE) == 0) break; } - vacuum_delay_point(); + vacuum_delay_point(false); next_unskippable_block++; } } @@ -1127,7 +1127,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) if ((vmskipflags & VISIBILITYMAP_ALL_VISIBLE) == 0) break; } - vacuum_delay_point(); + vacuum_delay_point(false); next_unskippable_block++; } } @@ -1177,7 +1177,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) all_visible_according_to_vm = true; } - vacuum_delay_point(); + vacuum_delay_point(false); /* * Regularly check if wraparound failsafe should trigger. @@ -2374,7 +2374,7 @@ lazy_vacuum_heap_rel(LVRelState *vacrel) Page page; Size freespace; - vacuum_delay_point(); + vacuum_delay_point(false); tblk = ItemPointerGetBlockNumber(&vacrel->dead_tuples->itemptrs[tupindex]); vacrel->blkno = tblk; diff --git a/src/backend/access/nbtree/nbtree.c b/src/backend/access/nbtree/nbtree.c index 8d4a587899c..d8e967402b2 100644 --- a/src/backend/access/nbtree/nbtree.c +++ b/src/backend/access/nbtree/nbtree.c @@ -1162,7 +1162,7 @@ btvacuumpage(BTVacState *vstate, BlockNumber scanblkno) backtrack_to = P_NONE; /* call vacuum_delay_point while not holding any buffer lock */ - vacuum_delay_point(); + vacuum_delay_point(false); /* * We can't use _bt_getbuf() here because it always applies diff --git a/src/backend/access/spgist/spgvacuum.c b/src/backend/access/spgist/spgvacuum.c index 76fb0374c42..9a87a00ae42 100644 --- a/src/backend/access/spgist/spgvacuum.c +++ b/src/backend/access/spgist/spgvacuum.c @@ -615,7 +615,7 @@ spgvacuumpage(spgBulkDeleteState *bds, BlockNumber blkno) Page page; /* call vacuum_delay_point while not holding any buffer lock */ - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, bds->info->strategy); @@ -694,7 +694,7 @@ spgprocesspending(spgBulkDeleteState *bds) continue; /* ignore already-done items */ /* call vacuum_delay_point while not holding any buffer lock */ - vacuum_delay_point(); + vacuum_delay_point(false); /* examine the referenced page */ blkno = ItemPointerGetBlockNumber(&pitem->tid); diff --git a/src/backend/commands/analyze.c b/src/backend/commands/analyze.c index 2630c4943f8..480e60c8507 100644 --- a/src/backend/commands/analyze.c +++ b/src/backend/commands/analyze.c @@ -1406,7 +1406,7 @@ compute_index_stats(Relation onerel, double totalrows, { HeapTuple heapTuple = rows[rowno]; - vacuum_delay_point(); + vacuum_delay_point(true); /* * Reset the per-tuple context each time, to reclaim any cruft @@ -1826,7 +1826,7 @@ acquire_sample_rows(Relation onerel, int elevel, prefetch_targblock = BlockSampler_Next(&prefetch_bs); #endif - vacuum_delay_point(); + vacuum_delay_point(true); block_accepted = table_scan_analyze_next_block(scan, targblock, vac_strategy); @@ -3452,7 +3452,7 @@ compute_trivial_stats(VacAttrStatsP stats, Datum value; bool isnull; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, i, &isnull); @@ -3574,7 +3574,7 @@ compute_distinct_stats(VacAttrStatsP stats, int firstcount1, j; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, i, &isnull); @@ -3934,7 +3934,7 @@ compute_scalar_stats(VacAttrStatsP stats, Datum value; bool isnull; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, i, &isnull); diff --git a/src/backend/commands/vacuum.c b/src/backend/commands/vacuum.c index 65cf10bb833..4335fb6afc9 100644 --- a/src/backend/commands/vacuum.c +++ b/src/backend/commands/vacuum.c @@ -2985,7 +2985,7 @@ vac_close_indexes(int nindexes, Relation *Irel, LOCKMODE lockmode) * typically once per page processed. */ void -vacuum_delay_point(void) +vacuum_delay_point(bool is_analyze) { double msec = 0; diff --git a/src/backend/tsearch/ts_typanalyze.c b/src/backend/tsearch/ts_typanalyze.c index 504ba1569ee..5c5ff6d8dc1 100644 --- a/src/backend/tsearch/ts_typanalyze.c +++ b/src/backend/tsearch/ts_typanalyze.c @@ -206,7 +206,7 @@ compute_tsvector_stats(VacAttrStats *stats, char *lexemesptr; int j; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, vector_no, &isnull); diff --git a/src/backend/utils/adt/array_typanalyze.c b/src/backend/utils/adt/array_typanalyze.c index 8993d23e18b..71f99bbd327 100644 --- a/src/backend/utils/adt/array_typanalyze.c +++ b/src/backend/utils/adt/array_typanalyze.c @@ -314,7 +314,7 @@ compute_array_stats(VacAttrStats *stats, AnalyzeAttrFetchFunc fetchfunc, int distinct_count; bool count_item_found; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, array_no, &isnull); if (isnull) diff --git a/src/backend/utils/adt/rangetypes_typanalyze.c b/src/backend/utils/adt/rangetypes_typanalyze.c index 9d5cf897c45..444c6c2da8e 100644 --- a/src/backend/utils/adt/rangetypes_typanalyze.c +++ b/src/backend/utils/adt/rangetypes_typanalyze.c @@ -168,7 +168,7 @@ compute_range_stats(VacAttrStats *stats, AnalyzeAttrFetchFunc fetchfunc, upper; float8 length; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, range_no, &isnull); if (isnull) diff --git a/src/include/commands/vacuum.h b/src/include/commands/vacuum.h index e33d39973c0..5d4309d927a 100644 --- a/src/include/commands/vacuum.h +++ b/src/include/commands/vacuum.h @@ -409,7 +409,7 @@ extern void vacuum_set_xid_limits(Relation rel, extern bool vacuum_xid_failsafe_check(TransactionId relfrozenxid, MultiXactId relminmxid); extern void vac_update_datfrozenxid(void); -extern void vacuum_delay_point(void); +extern void vacuum_delay_point(bool is_analyze); extern bool vacuum_is_relation_owner(Oid relid, Form_pg_class reltuple, bits32 options); extern Relation vacuum_open_relation(Oid relid, RangeVar *relation, From 25921b080f3f016331b85bffbb9a356a8c1b5f41 Mon Sep 17 00:00:00 2001 From: Nathan Bossart Date: Tue, 11 Feb 2025 16:38:14 -0600 Subject: [PATCH 02/18] Measure cost-based vacuum delay Introduce the process-local VacuumDelayTime accumulator and measure time spent in cost-based VACUUM delays, in microseconds. Gate the measurement with track_cost_delay_timing, disabled by default. Consumers take differences around their operations to attribute the measured delay. Use is_analyze to exclude ANALYZE sampling and statistics computation from the vacuum delay accumulator. Cost-based throttling and interrupt checks continue to apply to both operations. Port the delay measurement from PostgreSQL 18 without its progress-view changes, so this commit needs no catalog update. Omit the progress slots, progress-increment helper and parallel progress reporting. The accumulator is process-local and Cloudberry does not enable parallel vacuum. Register the GUC in the unsynchronized list. Author: Bertrand Drouvot Co-authored-by: Nathan Bossart Co-authored-by: Alena Rybakina Discussion: https://postgr.es/m/ZmaXmWDL829fzAVX%40ip-10-97-1-34.eu-west-3.compute.internal Backported from PostgreSQL 18 (commit bb8dff9995f2cf501376772898bcbcf58aa05cde). (cherry picked from commit 79c8bbcd33cccee0fde09041316ee5b1dc79c614) --- doc/src/sgml/config.sgml | 20 ++++++++++++++++ src/backend/commands/vacuum.c | 24 +++++++++++++++++++ src/backend/utils/misc/guc.c | 9 +++++++ src/backend/utils/misc/postgresql.conf.sample | 1 + src/include/commands/vacuum.h | 2 ++ src/include/utils/unsync_guc_name.h | 1 + 6 files changed, 57 insertions(+) diff --git a/doc/src/sgml/config.sgml b/doc/src/sgml/config.sgml index 0396704f3d8..ef4e256834b 100644 --- a/doc/src/sgml/config.sgml +++ b/doc/src/sgml/config.sgml @@ -7601,6 +7601,26 @@ COPY postgres_log FROM '/full/path/to/logfile.csv' WITH csv; + + track_cost_delay_timing (boolean) + + track_cost_delay_timing configuration parameter + + + + + Enables timing of cost-based vacuum delay (see + ). This parameter + is off by default, as it will repeatedly query the operating system for + the current time, which may cause significant overhead on some + platforms. You can use the tool to + measure the overhead of timing on your system. The measured time is + included in cumulative vacuum statistics. Only superusers can change + this setting. + + + + track_io_timing (boolean) diff --git a/src/backend/commands/vacuum.c b/src/backend/commands/vacuum.c index 4335fb6afc9..40b36c7b9bc 100644 --- a/src/backend/commands/vacuum.c +++ b/src/backend/commands/vacuum.c @@ -104,6 +104,15 @@ static MemoryContext vac_context = NULL; static BufferAccessStrategy vac_strategy; +/* + * Cumulative time this process has spent in cost-based VACUUM delays, in + * microseconds. Consumers take differences around an operation. ANALYZE + * uses the same delay function but does not contribute to this accumulator. + * Updated only while track_cost_delay_timing is enabled. + */ +int64 VacuumDelayTime = 0; +bool track_cost_delay_timing = false; + /* * Variables for cost-based parallel vacuum. See comments atop * compute_parallel_delay to understand how it works. @@ -3007,13 +3016,28 @@ vacuum_delay_point(bool is_analyze) /* Nap if appropriate */ if (msec > 0) { + instr_time delay_start; + instr_time delay_end; + if (msec > VacuumCostDelay * 4) msec = VacuumCostDelay * 4; pgstat_report_wait_start(WAIT_EVENT_VACUUM_DELAY); + if (track_cost_delay_timing && !is_analyze) + INSTR_TIME_SET_CURRENT(delay_start); pg_usleep(msec * 1000); pgstat_report_wait_end(); + if (track_cost_delay_timing && !is_analyze) + { + int64 delay_us; + + INSTR_TIME_SET_CURRENT(delay_end); + INSTR_TIME_SUBTRACT(delay_end, delay_start); + delay_us = (int64) INSTR_TIME_GET_MICROSEC(delay_end); + VacuumDelayTime += delay_us; + } + /* * We don't want to ignore postmaster death during very long vacuums * with vacuum_cost_delay configured. We can't use the usual diff --git a/src/backend/utils/misc/guc.c b/src/backend/utils/misc/guc.c index 6166c5ff249..2f8d498eb20 100644 --- a/src/backend/utils/misc/guc.c +++ b/src/backend/utils/misc/guc.c @@ -1627,6 +1627,15 @@ static struct config_bool ConfigureNamesBool[] = true, NULL, NULL, NULL }, + { + {"track_cost_delay_timing", PGC_SUSET, STATS_COLLECTOR, + gettext_noop("Collects timing statistics for cost-based vacuum delay."), + NULL + }, + &track_cost_delay_timing, + false, + NULL, NULL, NULL + }, { {"track_io_timing", PGC_SUSET, STATS_COLLECTOR, gettext_noop("Collects timing statistics for database I/O activity."), diff --git a/src/backend/utils/misc/postgresql.conf.sample b/src/backend/utils/misc/postgresql.conf.sample index 4192dfb2748..a5dc8e88745 100644 --- a/src/backend/utils/misc/postgresql.conf.sample +++ b/src/backend/utils/misc/postgresql.conf.sample @@ -626,6 +626,7 @@ optimizer_analyze_root_partition = on # stats collection on root partitions #track_activities = on #track_activity_query_size = 1024 # (change requires restart) #track_counts = off +#track_cost_delay_timing = off #track_io_timing = off #track_wal_io_timing = off #track_functions = none # none, pl, all diff --git a/src/include/commands/vacuum.h b/src/include/commands/vacuum.h index 5d4309d927a..7b80970f0cd 100644 --- a/src/include/commands/vacuum.h +++ b/src/include/commands/vacuum.h @@ -371,6 +371,8 @@ extern int vacuum_multixact_failsafe_age; extern pg_atomic_uint32 *VacuumSharedCostBalance; extern pg_atomic_uint32 *VacuumActiveNWorkers; extern int VacuumCostBalanceLocal; +extern PGDLLIMPORT int64 VacuumDelayTime; +extern PGDLLIMPORT bool track_cost_delay_timing; /* in commands/vacuum.c */ diff --git a/src/include/utils/unsync_guc_name.h b/src/include/utils/unsync_guc_name.h index 85ecb3548e6..4bfb7d5d06e 100644 --- a/src/include/utils/unsync_guc_name.h +++ b/src/include/utils/unsync_guc_name.h @@ -599,6 +599,7 @@ "track_activities", "track_activity_query_size", "track_commit_timestamp", + "track_cost_delay_timing", "track_counts", "track_functions", "track_io_timing", From 70be33c2b3e7b9ce086afcc04f0777793a2176d6 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Wed, 9 Sep 2026 14:49:54 +0300 Subject: [PATCH 03/18] Track table visibility map stability Collect visible_page_marks_cleared and frozen_page_marks_cleared. Count transitions of heap visibility-map bits from set to clear during normal backend activity. Further modifications of a page whose bit is already clear do not increment that counter. Count clearings even if the modifying transaction subsequently rolls back. Comparing the all-visible clearing rate with DML volume helps identify modifications spread over pages vacuum previously marked all-visible. Index-only scans must visit the heap for those pages until visibility is re-established. The all-frozen counter shows how often pages lose the property that allows aggressive vacuum to skip them. Clearing this bit does not unfreeze tuples that are already frozen. Adapt the v44 VM stability patch to Cloudberry's statistics collector. Deliver the counters with ordinary table-statistics messages and accumulate both relation and database totals. Collection follows track_counts. These are DML-driven visibility-map transitions rather than work performed by VACUUM. SQL access and TAP coverage follow in the separate vacuum_stats extension commit; this commit does not change the system catalog. Count the cleared bits in visibilitymap_clear(), which already takes a Relation on this branch. Skip relations without a statistics entry, including fake relations used during recovery. Authors: Alena Rybakina , Andrei Lepikhov (@danolivo), Andrei Zubkov (@zubkov-andrei) Based-on: https://www.postgresql.org/message-id/attachment/204710/v44-0009-Track-table-VM-stability.patch Co-authored-by: Andrei Lepikhov Co-authored-by: Andrei Zubkov --- src/backend/access/heap/visibilitymap.c | 15 ++++++++++++++ src/backend/postmaster/pgstat.c | 14 +++++++++++++ src/include/pgstat.h | 26 ++++++++++++++++++++++++- 3 files changed, 54 insertions(+), 1 deletion(-) diff --git a/src/backend/access/heap/visibilitymap.c b/src/backend/access/heap/visibilitymap.c index 7d252fa94ab..4663ade3aa3 100644 --- a/src/backend/access/heap/visibilitymap.c +++ b/src/backend/access/heap/visibilitymap.c @@ -90,6 +90,7 @@ #include "access/visibilitymap.h" #include "access/xlog.h" #include "miscadmin.h" +#include "pgstat.h" #include "port/pg_bitutils.h" #include "storage/bufmgr.h" #include "storage/lmgr.h" @@ -159,10 +160,24 @@ visibilitymap_clear(Relation rel, BlockNumber heapBlk, Buffer buf, uint8 flags) if (map[mapByte] & mask) { + uint8 cleared_bits = (map[mapByte] & mask) >> mapOffset; + map[mapByte] &= ~mask; MarkBufferDirty(buf); cleared = true; + + /* + * Count the pages that just lost their all-visible/all-frozen status + * for pg_stat_all_tables and pg_stat_database. + * The counters are delivered with the regular relation statistics, so + * nothing is counted during recovery, where rel is a fake relcache + * entry without a pgstat entry. + */ + if (cleared_bits & VISIBILITYMAP_ALL_VISIBLE) + pgstat_count_visible_page_marks_cleared(rel); + if (cleared_bits & VISIBILITYMAP_ALL_FROZEN) + pgstat_count_frozen_page_marks_cleared(rel); } LockBuffer(buf, BUFFER_LOCK_UNLOCK); diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c index 309a101fe02..9d7ba9fa55d 100644 --- a/src/backend/postmaster/pgstat.c +++ b/src/backend/postmaster/pgstat.c @@ -3820,6 +3820,8 @@ reset_dbentry_counters(PgStat_StatDBEntry *dbentry) dbentry->n_sessions_abandoned = 0; dbentry->n_sessions_fatal = 0; dbentry->n_sessions_killed = 0; + dbentry->n_frozen_page_marks_cleared = 0; + dbentry->n_visible_page_marks_cleared = 0; dbentry->stat_reset_timestamp = GetCurrentTimestamp(); dbentry->stats_timestamp = 0; @@ -3914,6 +3916,8 @@ pgstat_get_tab_entry(PgStat_StatDBEntry *dbentry, Oid tableoid, bool create) result->analyze_count = 0; result->autovac_analyze_timestamp = 0; result->autovac_analyze_count = 0; + result->frozen_page_marks_cleared = 0; + result->visible_page_marks_cleared = 0; } return result; @@ -5275,6 +5279,8 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) tabentry->analyze_count = 0; tabentry->autovac_analyze_timestamp = 0; tabentry->autovac_analyze_count = 0; + tabentry->frozen_page_marks_cleared = 0; + tabentry->visible_page_marks_cleared = 0; } else { @@ -5318,6 +5324,14 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) dbentry->n_tuples_deleted += tabmsg->t_counts.t_tuples_deleted; dbentry->n_blocks_fetched += tabmsg->t_counts.t_blocks_fetched; dbentry->n_blocks_hit += tabmsg->t_counts.t_blocks_hit; + tabentry->frozen_page_marks_cleared += + tabmsg->t_counts.t_frozen_page_marks_cleared; + tabentry->visible_page_marks_cleared += + tabmsg->t_counts.t_visible_page_marks_cleared; + dbentry->n_frozen_page_marks_cleared += + tabmsg->t_counts.t_frozen_page_marks_cleared; + dbentry->n_visible_page_marks_cleared += + tabmsg->t_counts.t_visible_page_marks_cleared; } } diff --git a/src/include/pgstat.h b/src/include/pgstat.h index c54b1bb369a..e9a5cff3af3 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -133,6 +133,9 @@ typedef struct PgStat_TableCounts PgStat_Counter t_blocks_fetched; PgStat_Counter t_blocks_hit; + + PgStat_Counter t_frozen_page_marks_cleared; + PgStat_Counter t_visible_page_marks_cleared; } PgStat_TableCounts; /* Possible targets for resetting cluster-wide shared values */ @@ -730,7 +733,7 @@ typedef union PgStat_Msg * ------------------------------------------------------------ */ -#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA2 +#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA3 /* ---------- * PgStat_StatDBEntry The collector's data per database @@ -769,6 +772,10 @@ typedef struct PgStat_StatDBEntry PgStat_Counter n_sessions_fatal; PgStat_Counter n_sessions_killed; + /* VM revisions are fed by ordinary relation statistics. */ + PgStat_Counter n_frozen_page_marks_cleared; + PgStat_Counter n_visible_page_marks_cleared; + TimestampTz stat_reset_timestamp; TimestampTz stats_timestamp; /* time of db stats file update */ @@ -816,6 +823,10 @@ typedef struct PgStat_StatTabEntry PgStat_Counter analyze_count; TimestampTz autovac_analyze_timestamp; /* autovacuum initiated */ PgStat_Counter autovac_analyze_count; + + /* VM revisions are fed by ordinary relation statistics. */ + PgStat_Counter frozen_page_marks_cleared; + PgStat_Counter visible_page_marks_cleared; } PgStat_StatTabEntry; @@ -1060,6 +1071,19 @@ extern void pgstat_report_connect(Oid dboid); extern void pgstat_report_autovac(Oid dboid); extern void pgstat_report_vacuum(Oid tableoid, bool shared, PgStat_Counter livetuples, PgStat_Counter deadtuples); + +/* count a page whose all-visible bit is being cleared */ +#define pgstat_count_visible_page_marks_cleared(rel) \ + do { \ + if ((rel)->pgstat_info != NULL) \ + (rel)->pgstat_info->t_counts.t_visible_page_marks_cleared++; \ + } while (0) +/* count a page whose all-frozen bit is being cleared */ +#define pgstat_count_frozen_page_marks_cleared(rel) \ + do { \ + if ((rel)->pgstat_info != NULL) \ + (rel)->pgstat_info->t_counts.t_frozen_page_marks_cleared++; \ + } while (0) extern void pgstat_report_analyze(Relation rel, PgStat_Counter livetuples, PgStat_Counter deadtuples, bool resetcounter); From 9fa477370fdd67b268f448bec3610a845d98ebdb Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Wed, 9 Sep 2026 14:49:54 +0300 Subject: [PATCH 04/18] Measure heap and index vacuum work per call Assemble backend-local tuple and page measurements and take per-call snapshots of the existing index results. Access methods accumulate results across passes, so retain each pass's delta without counting earlier work again. Keep SP-GiST's newly deleted page count cumulative within one vacuum, without counting pages that were already empty on entry. Assigning all deleted pages to pages_newly_deleted would count empty pages again on subsequent scans. --- src/backend/access/heap/vacuumlazy.c | 74 +++++++++++++++++++++++++++ src/backend/access/spgist/spgvacuum.c | 5 +- src/include/pgstat.h | 12 +++++ 3 files changed, 90 insertions(+), 1 deletion(-) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 76758799b22..5309be472fc 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -753,6 +753,20 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, rel->rd_rel->relisshared, Max(new_live_tuples, 0), vacrel->new_dead_tuples); + + /* assemble the per-vacuum measurements for subsequent reporting */ + { + PgStat_VacuumStats vacstats; + + MemSet(&vacstats, 0, sizeof(vacstats)); + vacstats.tuples_deleted = (PgStat_Counter) vacrel->tuples_deleted; + vacstats.dead_tuples = (PgStat_Counter) vacrel->new_dead_tuples; + vacstats.pages_deleted = (PgStat_Counter) vacrel->pages_removed; + + + + } + pgstat_progress_end_command(); /* and log the action if appropriate */ @@ -3044,6 +3058,60 @@ lazy_cleanup_all_indexes(LVRelState *vacrel) } } +/* + * lazy_index_vacstats_start() -- remember where an index vacuum call starts, + * for lazy_index_vacstats_finish(). + * + * The counters in istat accumulate over all the calls made for the index + * during one vacuum, so we report what each call added to them. That keeps + * the work of the bulk deletion passes accounted for even when the cleanup + * is skipped, and keeps it from being counted twice when it is not. + */ +static IndexBulkDeleteResult +lazy_index_vacstats_start(IndexBulkDeleteResult *istat) +{ + IndexBulkDeleteResult before; + + if (istat) + before = *istat; + else + MemSet(&before, 0, sizeof(before)); + + return before; +} + +/* + * lazy_index_vacstats_finish() -- finish measuring one index vacuum call. + * + * pages_deleted and pages_free describe the whole index as the call left it, + * not what it did, so they are only looked at after the cleanup, which comes + * last: the deleted pages that are not reusable yet are the index's dead + * pages. + */ +static void +lazy_index_vacstats_finish(Relation indrel, IndexBulkDeleteResult *istat, + IndexBulkDeleteResult *before, bool cleanup) +{ + PgStat_VacuumStats vacstats; + + MemSet(&vacstats, 0, sizeof(vacstats)); + if (istat) + { + vacstats.tuples_deleted = + (PgStat_Counter) (istat->tuples_removed - before->tuples_removed); + /* + * Access methods accumulate pages_newly_deleted over the calls, + * so report only the work performed by this call. + */ + if (istat->pages_newly_deleted >= before->pages_newly_deleted) + vacstats.pages_deleted = (PgStat_Counter) + (istat->pages_newly_deleted - before->pages_newly_deleted); + else + vacstats.pages_deleted = (PgStat_Counter) istat->pages_newly_deleted; + } + +} + /* * lazy_vacuum_one_index() -- vacuum index relation. * @@ -3062,6 +3130,7 @@ lazy_vacuum_one_index(Relation indrel, IndexBulkDeleteResult *istat, IndexVacuumInfo ivinfo; PGRUsage ru0; LVSavedErrInfo saved_err_info; + IndexBulkDeleteResult istat_before; pg_rusage_init(&ru0); @@ -3086,8 +3155,10 @@ lazy_vacuum_one_index(Relation indrel, IndexBulkDeleteResult *istat, InvalidBlockNumber, InvalidOffsetNumber); /* Do bulk deletion */ + istat_before = lazy_index_vacstats_start(istat); istat = index_bulk_delete(&ivinfo, istat, lazy_tid_reaped, (void *) vacrel->dead_tuples); + lazy_index_vacstats_finish(indrel, istat, &istat_before, false); ereport(elevel, (errmsg("scanned index \"%s\" to remove %d row versions", @@ -3118,6 +3189,7 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat, IndexVacuumInfo ivinfo; PGRUsage ru0; LVSavedErrInfo saved_err_info; + IndexBulkDeleteResult istat_before; pg_rusage_init(&ru0); @@ -3142,7 +3214,9 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat, VACUUM_ERRCB_PHASE_INDEX_CLEANUP, InvalidBlockNumber, InvalidOffsetNumber); + istat_before = lazy_index_vacstats_start(istat); istat = index_vacuum_cleanup(&ivinfo, istat); + lazy_index_vacstats_finish(indrel, istat, &istat_before, true); if (istat) { diff --git a/src/backend/access/spgist/spgvacuum.c b/src/backend/access/spgist/spgvacuum.c index 9a87a00ae42..8188dbedce0 100644 --- a/src/backend/access/spgist/spgvacuum.c +++ b/src/backend/access/spgist/spgvacuum.c @@ -613,6 +613,7 @@ spgvacuumpage(spgBulkDeleteState *bds, BlockNumber blkno) Relation index = bds->info->index; Buffer buffer; Page page; + bool was_empty; /* call vacuum_delay_point while not holding any buffer lock */ vacuum_delay_point(false); @@ -621,6 +622,7 @@ spgvacuumpage(spgBulkDeleteState *bds, BlockNumber blkno) RBM_NORMAL, bds->info->strategy); LockBuffer(buffer, BUFFER_LOCK_EXCLUSIVE); page = (Page) BufferGetPage(buffer); + was_empty = PageIsNew(page) || PageIsEmpty(page); if (PageIsNew(page)) { @@ -664,6 +666,8 @@ spgvacuumpage(spgBulkDeleteState *bds, BlockNumber blkno) { RecordFreeIndexPage(index, blkno); bds->stats->pages_deleted++; + if (!was_empty) + bds->stats->pages_newly_deleted++; } else { @@ -891,7 +895,6 @@ spgvacuumscan(spgBulkDeleteState *bds) /* Report final stats */ bds->stats->num_pages = num_pages; - bds->stats->pages_newly_deleted = bds->stats->pages_deleted; bds->stats->pages_free = bds->stats->pages_deleted; } diff --git a/src/include/pgstat.h b/src/include/pgstat.h index e9a5cff3af3..804cfc5e4f2 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -422,6 +422,18 @@ typedef struct PgStat_MsgVacuum } PgStat_MsgVacuum; +/* ---------- + * PgStat_VacuumStats Vacuum statistics reported for a relation. + * ---------- + */ +typedef struct PgStat_VacuumStats +{ + PgStat_Counter tuples_deleted; /* tuples removed by vacuum */ + PgStat_Counter dead_tuples; /* dead tuples left unremoved */ + PgStat_Counter pages_deleted; /* pages removed/deleted by vacuum */ + +} PgStat_VacuumStats; + /* ---------- * PgStat_MsgAnalyze Sent by the backend or autovacuum daemon * after ANALYZE From d60eb23c47a8c203f04286a09837d88306eceb29 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Wed, 9 Sep 2026 14:49:54 +0300 Subject: [PATCH 05/18] Count vacuum pages with unremovable dead tuples Introduce dead_pages in LVRelState and PgStat_VacuumStats. Increment it for each heap page that retains dead tuples after pruning. For index cleanup, derive the count from deleted pages not yet reusable. Collector delivery and SQL exposure follow in the reporting commit. Include these per-operation measurements in VACUUM VERBOSE output. --- src/backend/access/heap/vacuumlazy.c | 14 ++++++++++++++ src/include/pgstat.h | 1 + 2 files changed, 15 insertions(+) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 5309be472fc..31686979c96 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -370,6 +370,8 @@ typedef struct LVRelState BlockNumber pages_removed; /* pages remove by truncation */ BlockNumber lpdead_item_pages; /* # pages with LP_DEAD items */ BlockNumber nonempty_pages; /* actually, last nonempty page + 1 */ + /* Counters reported as the relation's vacuum statistics */ + BlockNumber dead_pages; /* pages left with unremovable dead tuples */ /* Statistics output by us, for table */ double new_rel_tuples; /* new estimated total # of tuples */ @@ -762,6 +764,7 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacstats.tuples_deleted = (PgStat_Counter) vacrel->tuples_deleted; vacstats.dead_tuples = (PgStat_Counter) vacrel->new_dead_tuples; vacstats.pages_deleted = (PgStat_Counter) vacrel->pages_removed; + vacstats.dead_pages = (PgStat_Counter) vacrel->dead_pages; @@ -832,6 +835,8 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacrel->rel_pages, vacrel->pinskipped_pages, vacrel->frozenskipped_pages); + appendStringInfo(&buf, _("pages with dead tuples not yet removable: %u\n"), + vacrel->dead_pages); appendStringInfo(&buf, _("tuples: %lld removed, %lld remain, %lld are dead but not yet removable, oldest xmin: %u\n"), (long long) vacrel->tuples_deleted, @@ -1699,6 +1704,8 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) appendStringInfo(&buf, _("%lld dead row versions cannot be removed yet, oldest xmin: %u\n"), (long long) vacrel->new_dead_tuples, vacrel->OldestXmin); + appendStringInfo(&buf, _("pages with dead tuples not yet removable: %u\n"), + vacrel->dead_pages); appendStringInfo(&buf, ngettext("Skipped %u page due to buffer pins, ", "Skipped %u pages due to buffer pins, ", vacrel->pinskipped_pages), @@ -2106,6 +2113,10 @@ lazy_scan_prune(LVRelState *vacrel, dead_tuples->num_tuples); } + /* Remember pages that keep dead tuples we could not remove yet */ + if (new_dead_tuples > 0) + vacrel->dead_pages++; + /* Finally, add page-local counts to whole-VACUUM counts */ vacrel->tuples_deleted += tuples_deleted; vacrel->lpdead_items += lpdead_items; @@ -3108,6 +3119,9 @@ lazy_index_vacstats_finish(Relation indrel, IndexBulkDeleteResult *istat, (istat->pages_newly_deleted - before->pages_newly_deleted); else vacstats.pages_deleted = (PgStat_Counter) istat->pages_newly_deleted; + if (cleanup && istat->pages_deleted > istat->pages_free) + vacstats.dead_pages = + (PgStat_Counter) (istat->pages_deleted - istat->pages_free); } } diff --git a/src/include/pgstat.h b/src/include/pgstat.h index 804cfc5e4f2..6bb3725b131 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -431,6 +431,7 @@ typedef struct PgStat_VacuumStats PgStat_Counter tuples_deleted; /* tuples removed by vacuum */ PgStat_Counter dead_tuples; /* dead tuples left unremoved */ PgStat_Counter pages_deleted; /* pages removed/deleted by vacuum */ + PgStat_Counter dead_pages; /* pages with unremoved dead tuples */ } PgStat_VacuumStats; From b6ea192ce7abc77f66e5e3e66114bd306475cbdd Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Wed, 9 Sep 2026 14:49:54 +0300 Subject: [PATCH 06/18] Count heap pages where vacuum freezes tuples Introduce pages_frozen in LVRelState and PgStat_VacuumStats. Increment the counter when vacuum executes freezing work on a heap page and include it in the backend-local measurements. Collector reporting and SQL exposure follow separately. Include these per-operation measurements in VACUUM VERBOSE output. --- src/backend/access/heap/vacuumlazy.c | 8 ++++++++ src/include/pgstat.h | 1 + 2 files changed, 9 insertions(+) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 31686979c96..8b09aab35ca 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -372,6 +372,7 @@ typedef struct LVRelState BlockNumber nonempty_pages; /* actually, last nonempty page + 1 */ /* Counters reported as the relation's vacuum statistics */ BlockNumber dead_pages; /* pages left with unremovable dead tuples */ + BlockNumber pages_frozen; /* pages where we froze tuples */ /* Statistics output by us, for table */ double new_rel_tuples; /* new estimated total # of tuples */ @@ -765,6 +766,7 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacstats.dead_tuples = (PgStat_Counter) vacrel->new_dead_tuples; vacstats.pages_deleted = (PgStat_Counter) vacrel->pages_removed; vacstats.dead_pages = (PgStat_Counter) vacrel->dead_pages; + vacstats.pages_frozen = (PgStat_Counter) vacrel->pages_frozen; @@ -837,6 +839,8 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacrel->frozenskipped_pages); appendStringInfo(&buf, _("pages with dead tuples not yet removable: %u\n"), vacrel->dead_pages); + appendStringInfo(&buf, _("pages with tuples frozen: %u\n"), + vacrel->pages_frozen); appendStringInfo(&buf, _("tuples: %lld removed, %lld remain, %lld are dead but not yet removable, oldest xmin: %u\n"), (long long) vacrel->tuples_deleted, @@ -1706,6 +1710,8 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) (long long) vacrel->new_dead_tuples, vacrel->OldestXmin); appendStringInfo(&buf, _("pages with dead tuples not yet removable: %u\n"), vacrel->dead_pages); + appendStringInfo(&buf, _("pages with tuples frozen: %u\n"), + vacrel->pages_frozen); appendStringInfo(&buf, ngettext("Skipped %u page due to buffer pins, ", "Skipped %u pages due to buffer pins, ", vacrel->pinskipped_pages), @@ -2013,6 +2019,8 @@ lazy_scan_prune(LVRelState *vacrel, { Assert(prunestate->hastup); + vacrel->pages_frozen++; + /* * At least one tuple with storage needs to be frozen -- execute that * now. diff --git a/src/include/pgstat.h b/src/include/pgstat.h index 6bb3725b131..9720a0ea803 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -432,6 +432,7 @@ typedef struct PgStat_VacuumStats PgStat_Counter dead_tuples; /* dead tuples left unremoved */ PgStat_Counter pages_deleted; /* pages removed/deleted by vacuum */ PgStat_Counter dead_pages; /* pages with unremoved dead tuples */ + PgStat_Counter pages_frozen; /* pages where vacuum froze tuples */ } PgStat_VacuumStats; From 4d1d4adcb5cd82de234b0ae74fd0bef368c2baee Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Wed, 9 Sep 2026 14:49:54 +0300 Subject: [PATCH 07/18] Count heap pages marked all-visible by vacuum Introduce pages_all_visible in LVRelState and PgStat_VacuumStats. Count visibility-map updates for empty pages, scanned pages and pages revisited during heap cleanup, testing the all-visible bit when needed. Collector delivery and SQL exposure follow in the reporting commit. Include these per-operation measurements in VACUUM VERBOSE output. Related to vm_new_visible_pages in v40-0006, "Extended vacuum statistics: visibility-map page transitions for tables". That patch reports existing PostgreSQL VM transition counters; here the all-visible measurement is added to Cloudberry's PostgreSQL 14 vacuum paths. The patch's separate all-frozen and combined visible/frozen counters are not included here. Related-to: https://www.postgresql.org/message-id/attachment/199749/v40-0006-Extended-vacuum-statistics-visibility-map-page-trans.patch --- src/backend/access/heap/vacuumlazy.c | 12 ++++++++++++ src/include/pgstat.h | 1 + 2 files changed, 13 insertions(+) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 8b09aab35ca..435287cbc87 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -373,6 +373,7 @@ typedef struct LVRelState /* Counters reported as the relation's vacuum statistics */ BlockNumber dead_pages; /* pages left with unremovable dead tuples */ BlockNumber pages_frozen; /* pages where we froze tuples */ + BlockNumber pages_all_visible; /* pages we marked all-visible */ /* Statistics output by us, for table */ double new_rel_tuples; /* new estimated total # of tuples */ @@ -767,6 +768,7 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacstats.pages_deleted = (PgStat_Counter) vacrel->pages_removed; vacstats.dead_pages = (PgStat_Counter) vacrel->dead_pages; vacstats.pages_frozen = (PgStat_Counter) vacrel->pages_frozen; + vacstats.pages_all_visible = (PgStat_Counter) vacrel->pages_all_visible; @@ -841,6 +843,8 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacrel->dead_pages); appendStringInfo(&buf, _("pages with tuples frozen: %u\n"), vacrel->pages_frozen); + appendStringInfo(&buf, _("pages marked all-visible: %u\n"), + vacrel->pages_all_visible); appendStringInfo(&buf, _("tuples: %lld removed, %lld remain, %lld are dead but not yet removable, oldest xmin: %u\n"), (long long) vacrel->tuples_deleted, @@ -1414,6 +1418,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) visibilitymap_set(vacrel->rel, blkno, buf, InvalidXLogRecPtr, vmbuffer, InvalidTransactionId, VISIBILITYMAP_ALL_VISIBLE | VISIBILITYMAP_ALL_FROZEN); + vacrel->pages_all_visible++; END_CRIT_SECTION(); } @@ -1525,6 +1530,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) visibilitymap_set(vacrel->rel, blkno, buf, InvalidXLogRecPtr, vmbuffer, prunestate.visibility_cutoff_xid, flags); + vacrel->pages_all_visible++; } /* @@ -1712,6 +1718,8 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) vacrel->dead_pages); appendStringInfo(&buf, _("pages with tuples frozen: %u\n"), vacrel->pages_frozen); + appendStringInfo(&buf, _("pages marked all-visible: %u\n"), + vacrel->pages_all_visible); appendStringInfo(&buf, ngettext("Skipped %u page due to buffer pins, ", "Skipped %u pages due to buffer pins, ", vacrel->pinskipped_pages), @@ -2577,8 +2585,12 @@ lazy_vacuum_heap_page(LVRelState *vacrel, BlockNumber blkno, Buffer buffer, Assert(BufferIsValid(*vmbuffer)); if (flags != 0) + { visibilitymap_set(vacrel->rel, blkno, buffer, InvalidXLogRecPtr, *vmbuffer, visibility_cutoff_xid, flags); + if (flags & VISIBILITYMAP_ALL_VISIBLE) + vacrel->pages_all_visible++; + } } /* Revert to the previous phase information for error traceback */ diff --git a/src/include/pgstat.h b/src/include/pgstat.h index 9720a0ea803..eb74dd892b8 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -433,6 +433,7 @@ typedef struct PgStat_VacuumStats PgStat_Counter pages_deleted; /* pages removed/deleted by vacuum */ PgStat_Counter dead_pages; /* pages with unremoved dead tuples */ PgStat_Counter pages_frozen; /* pages where vacuum froze tuples */ + PgStat_Counter pages_all_visible; /* pages marked all-visible by vacuum */ } PgStat_VacuumStats; From f808f0eb129987bbec6edad79c124f30fb97de57 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Wed, 9 Sep 2026 14:49:54 +0300 Subject: [PATCH 08/18] Count vacuum runs made aggressive by the freeze age Introduce freeze_age_vacuum_count and preserve the freeze-age decision before DISABLE_PAGE_SKIPPING can make the scan aggressive. Record one for a run driven by relfrozenxid or relminmxid, and zero when aggressive scanning is forced only by DISABLE_PAGE_SKIPPING. Include VACUUM FREEZE, which sets the freeze-age thresholds to zero. This introduces the measurement; collector accumulation and SQL exposure follow in the reporting commit. Include these per-operation measurements in VACUUM VERBOSE output. --- src/backend/access/heap/vacuumlazy.c | 17 ++++++++++++++++- src/include/pgstat.h | 7 +++++++ 2 files changed, 23 insertions(+), 1 deletion(-) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 435287cbc87..b1b6715ad89 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -516,6 +516,7 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, write_rate; bool aggressive; /* should we scan all unfrozen pages? */ bool scanned_all_unfrozen; /* actually scanned all such pages? */ + bool freeze_age_vacuum; /* aggressive due to freeze age? */ char **indnames = NULL; TransactionId xidFullScanLimit; MultiXactId mxactFullScanLimit; @@ -578,6 +579,14 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, xidFullScanLimit); aggressive |= MultiXactIdPrecedesOrEquals(rel->rd_rel->relminmxid, mxactFullScanLimit); + + /* + * Remember whether the freeze table age made this run aggressive. + * DISABLE_PAGE_SKIPPING can also force an aggressive scan, but does + * not by itself contribute to freeze_age_vacuum_count. + */ + freeze_age_vacuum = aggressive; + if (params->options & VACOPT_DISABLE_PAGE_SKIPPING) aggressive = true; @@ -769,8 +778,12 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacstats.dead_pages = (PgStat_Counter) vacrel->dead_pages; vacstats.pages_frozen = (PgStat_Counter) vacrel->pages_frozen; vacstats.pages_all_visible = (PgStat_Counter) vacrel->pages_all_visible; + vacstats.freeze_age_vacuum_count = freeze_age_vacuum ? 1 : 0; - + ereport(elevel, + (errmsg("table \"%s\": vacuum statistics", vacrel->relname), + errdetail("aggressive scan required by freeze age: %s", + freeze_age_vacuum ? _("yes") : _("no")))); } @@ -845,6 +858,8 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacrel->pages_frozen); appendStringInfo(&buf, _("pages marked all-visible: %u\n"), vacrel->pages_all_visible); + appendStringInfo(&buf, _("aggressive scan required by freeze age: %s\n"), + freeze_age_vacuum ? _("yes") : _("no")); appendStringInfo(&buf, _("tuples: %lld removed, %lld remain, %lld are dead but not yet removable, oldest xmin: %u\n"), (long long) vacrel->tuples_deleted, diff --git a/src/include/pgstat.h b/src/include/pgstat.h index eb74dd892b8..fd9ba381194 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -435,6 +435,13 @@ typedef struct PgStat_VacuumStats PgStat_Counter pages_frozen; /* pages where vacuum froze tuples */ PgStat_Counter pages_all_visible; /* pages marked all-visible by vacuum */ + /* + * Number of heap vacuum runs made aggressive by the XID or MultiXact + * freeze table age. This includes VACUUM FREEZE, which sets those + * age thresholds to zero, but not DISABLE_PAGE_SKIPPING alone. + */ + PgStat_Counter freeze_age_vacuum_count; + } PgStat_VacuumStats; /* ---------- From 37abafc9513d5f280ac5ca9368efaad209ddd8de Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Mon, 21 Sep 2026 01:40:05 +0300 Subject: [PATCH 09/18] Measure append-optimized vacuum phases and remaining dead tuples Preserve remaining dead tuples from post-cleanup and accumulate elapsed and cost-delay time across AO vacuum phases. Measure index cleanup time as well. Widen existing compaction tuple and byte accumulators to int64, and keep the reclaimed-block conversion 64-bit as well. Collector reporting and SQL-level AO tests follow in a separate commit. Include these per-operation measurements in VACUUM VERBOSE output. Measure cleanup-only index scans when there are no obsolete AO segments. --- src/backend/access/appendonly/aomd.c | 3 +- src/backend/commands/vacuum_ao.c | 68 ++++++++++++++++++++-- src/include/access/appendonly_compaction.h | 9 ++- 3 files changed, 73 insertions(+), 7 deletions(-) diff --git a/src/backend/access/appendonly/aomd.c b/src/backend/access/appendonly/aomd.c index 342e7771b7b..36dfa5d71f5 100644 --- a/src/backend/access/appendonly/aomd.c +++ b/src/backend/access/appendonly/aomd.c @@ -221,7 +221,8 @@ TruncateAOSegmentFile(File fd, Relation rel, int32 segFileNum, int64 offset, AOV /* report heap-equivalent blocks vacuumed */ vacrelstats->nbytes_truncated += filesize_before - offset; pgstat_progress_update_param(PROGRESS_VACUUM_HEAP_BLKS_VACUUMED, - RelationGuessNumberOfBlocksFromSize(vacrelstats->nbytes_truncated)); + vacrelstats->nbytes_truncated / BLCKSZ + + (vacrelstats->nbytes_truncated % BLCKSZ != 0)); } if (XLogIsNeeded() && RelationNeedsWAL(rel)) diff --git a/src/backend/commands/vacuum_ao.c b/src/backend/commands/vacuum_ao.c index dff6ecf332d..f41feed8226 100644 --- a/src/backend/commands/vacuum_ao.c +++ b/src/backend/commands/vacuum_ao.c @@ -323,6 +323,12 @@ ao_vacuum_rel_post_cleanup(Relation onerel, VacuumParams *params, BufferAccessSt reltuples, deadtuples); + /* + * Remember what is left behind for the vacuum statistics, which + * ao_vacuum_rel() reports once this last phase is over. + */ + vacrelstats->dead_tuples_left = (int64) deadtuples; + SIMPLE_FAULT_INJECTOR("vacuum_ao_post_cleanup_end"); } @@ -443,6 +449,10 @@ void ao_vacuum_rel(Relation rel, VacuumParams *params, BufferAccessStrategy bstrategy) { static AOVacuumRelStats *vacrelstats = NULL; + instr_time phasestart; + instr_time phaseend; + int64 startdelaytime; + Assert(RelationStorageIsAO(rel)); Assert(params != NULL); @@ -469,20 +479,50 @@ ao_vacuum_rel(Relation rel, VacuumParams *params, BufferAccessStrategy bstrategy /* * Do the actual work --- either FULL or "lazy" vacuum + * + * Each phase is timed for the vacuum statistics, which are reported for + * the relation once the last phase is done. The phases run in separate + * transactions and may even end up in different vacuum workers; when that + * happens vacrelstats is reset above, and the statistics report describes + * the phases this worker did. */ + INSTR_TIME_SET_CURRENT(phasestart); + startdelaytime = VacuumDelayTime; + if (ao_vacuum_phase == VACOPT_AO_PRE_CLEANUP_PHASE) ao_vacuum_rel_pre_cleanup(rel, params, bstrategy, vacrelstats); else if (ao_vacuum_phase == VACOPT_AO_COMPACT_PHASE) ao_vacuum_rel_compact(rel, params, bstrategy, vacrelstats); else if (ao_vacuum_phase == VACOPT_AO_POST_CLEANUP_PHASE) - { ao_vacuum_rel_post_cleanup(rel, params, bstrategy, vacrelstats); - pgstat_progress_end_command(); - cleanup_vacrelstats(&vacrelstats); - } else /* Do nothing here, we will launch the stages later */ Assert(ao_vacuum_phase == 0); + + INSTR_TIME_SET_CURRENT(phaseend); + INSTR_TIME_SUBTRACT(phaseend, phasestart); + vacrelstats->vacuum_time += (int64) INSTR_TIME_GET_MICROSEC(phaseend); + vacrelstats->delay_time += VacuumDelayTime - startdelaytime; + + if (ao_vacuum_phase == VACOPT_AO_POST_CLEANUP_PHASE) + { + int elevel = (params->options & VACOPT_VERBOSE) ? INFO : DEBUG2; + + if (Gp_role == GP_ROLE_DISPATCH) + elevel = DEBUG2; + + ereport(elevel, + (errmsg("append-optimized table \"%s\": vacuum statistics", + RelationGetRelationName(rel)), + errdetail("%lld dead tuples remain.\n" + "elapsed: %.3f ms, cost-based delay: %.3f ms", + (long long) vacrelstats->dead_tuples_left, + vacrelstats->vacuum_time / 1000.0, + vacrelstats->delay_time / 1000.0))); + + pgstat_progress_end_command(); + cleanup_vacrelstats(&vacrelstats); + } } /* @@ -633,10 +673,15 @@ vacuum_appendonly_index(Relation indexRelation, IndexBulkDeleteResult *stats; IndexVacuumInfo ivinfo = {0}; PGRUsage ru0; + instr_time starttime; + instr_time endtime; + int64 startdelaytime; Assert(RelationIsValid(indexRelation)); pg_rusage_init(&ru0); + INSTR_TIME_SET_CURRENT(starttime); + startdelaytime = VacuumDelayTime; ivinfo.index = indexRelation; ivinfo.analyze_only = false; @@ -661,6 +706,9 @@ vacuum_appendonly_index(Relation indexRelation, /* Do post-VACUUM cleanup */ stats = index_vacuum_cleanup(&ivinfo, stats); + INSTR_TIME_SET_CURRENT(endtime); + INSTR_TIME_SUBTRACT(endtime, starttime); + if (!stats) return; @@ -685,9 +733,11 @@ vacuum_appendonly_index(Relation indexRelation, stats->num_pages), errdetail("%.0f index row versions were removed.\n" "%u index pages have been deleted, %u are currently reusable.\n" + "cost-based delay: %.3f ms\n" "%s.", stats->tuples_removed, stats->pages_deleted, stats->pages_free, + (VacuumDelayTime - startdelaytime) / 1000.0, pg_rusage_show(&ru0)))); pfree(stats); @@ -806,8 +856,13 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac IndexBulkDeleteResult *stats; IndexVacuumInfo ivinfo = {0}; PGRUsage ru0; + instr_time starttime; + instr_time endtime; + int64 startdelaytime; pg_rusage_init(&ru0); + INSTR_TIME_SET_CURRENT(starttime); + startdelaytime = VacuumDelayTime; ivinfo.index = indrel; ivinfo.analyze_only = false; @@ -824,6 +879,9 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac /* Do post-VACUUM cleanup */ stats = index_vacuum_cleanup(&ivinfo, NULL); + INSTR_TIME_SET_CURRENT(endtime); + INSTR_TIME_SUBTRACT(endtime, starttime); + if (!stats) return; @@ -847,8 +905,10 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac stats->num_index_tuples, stats->num_pages), errdetail("%u index pages have been deleted, %u are currently reusable.\n" + "cost-based delay: %.3f ms\n" "%s.", stats->pages_deleted, stats->pages_free, + (VacuumDelayTime - startdelaytime) / 1000.0, pg_rusage_show(&ru0)))); pfree(stats); diff --git a/src/include/access/appendonly_compaction.h b/src/include/access/appendonly_compaction.h index 44aa78a39fb..fcce48719aa 100644 --- a/src/include/access/appendonly_compaction.h +++ b/src/include/access/appendonly_compaction.h @@ -28,9 +28,14 @@ */ typedef struct AOVacuumRelStats { - int nbytes_truncated; /* current # of bytes truncated from segment file */ - int num_dead_tuples; /* current # of dead tuples */ + int64 nbytes_truncated; /* current # of bytes truncated from segment file */ + int64 num_dead_tuples; /* current # of dead tuples */ int num_index_vacuumed; /* current # of indexes been vacuumed */ + + /* for the vacuum statistics, accumulated over all the phases */ + int64 vacuum_time; /* time spent in the phases, in microseconds */ + int64 delay_time; /* of which the cost-based vacuum delay */ + int64 dead_tuples_left; /* tuples the post-cleanup found still hidden */ } AOVacuumRelStats; extern Bitmapset *AppendOptimizedCollectDeadSegments(Relation aorel); From 6c07d5af8fda6c58682b511a8ce7f00783dc849a Mon Sep 17 00:00:00 2001 From: Michael Paquier Date: Tue, 28 Jan 2025 09:57:32 +0900 Subject: [PATCH 10/18] Track per-relation cumulative time spent in [auto]vacuum and [auto]analyze This commit adds four fields to the statistics of relations, aggregating the amount of time spent for each operation on a relation: - total_vacuum_time, for manual vacuum. - total_autovacuum_time, for vacuum done by the autovacuum daemon. - total_analyze_time, for manual analyze. - total_autoanalyze_time, for analyze done by the autovacuum daemon. This gives users the option to derive the average time spent for these operations with the help of the related "count" fields. Bump PGSTAT_FILE_FORMAT_ID for the additions in PgStat_StatTabEntry. SQL access follows in the separate vacuum_stats extension commit; the system catalog is unchanged. Author: Sami Imseih Reviewed-by: Bertrand Drouvot, Michael Paquier Discussion: https://postgr.es/m/CAA5RZ0uVOGBYmPEeGF2d1B_67tgNjKx_bKDuL+oUftuoz+=Y1g@mail.gmail.com (cherry picked from commit 30a6ed0ce4bb18212ec38cdb537ea4b43bc99b83) (cherry picked from commit 6a99f9d44d89ad3c4219437e324d578962590053) (cherry picked from commit 5f76c5e42340226cdb4c8f1e3e1c549c65f9c560) Adaptation for this branch: carry elapsed time in UDP VACUUM/ANALYZE messages and initialize the new counters in both relation-entry creation paths. Keep microsecond precision internally and expose milliseconds. Time AO vacuum from the first phase executed by this worker and pass that recorded start time when reporting the final phase. --- src/backend/access/heap/vacuumlazy.c | 6 ++++-- src/backend/commands/analyze.c | 9 +++++---- src/backend/commands/vacuum_ao.c | 6 +++++- src/backend/postmaster/pgstat.c | 19 +++++++++++++++++-- src/include/access/appendonly_compaction.h | 2 ++ src/include/pgstat.h | 15 ++++++++++++--- 6 files changed, 45 insertions(+), 12 deletions(-) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index b1b6715ad89..04028f734e0 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -532,11 +532,13 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, TransactionId FreezeLimit; MultiXactId MultiXactCutoff; + /* Used for instrumentation and cumulative maintenance statistics. */ + starttime = GetCurrentTimestamp(); + /* measure elapsed time iff autovacuum logging requires it */ if (IsAutoVacuumWorkerProcess() && params->log_min_duration >= 0) { pg_rusage_init(&ru0); - starttime = GetCurrentTimestamp(); if (track_io_timing) { startreadtime = pgStatBlockReadTime; @@ -765,7 +767,7 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, pgstat_report_vacuum(RelationGetRelid(rel), rel->rd_rel->relisshared, Max(new_live_tuples, 0), - vacrel->new_dead_tuples); + vacrel->new_dead_tuples, starttime); /* assemble the per-vacuum measurements for subsequent reporting */ { diff --git a/src/backend/commands/analyze.c b/src/backend/commands/analyze.c index 480e60c8507..c1026843596 100644 --- a/src/backend/commands/analyze.c +++ b/src/backend/commands/analyze.c @@ -539,6 +539,9 @@ do_analyze_rel(Relation onerel, VacuumParams *params, save_sec_context | SECURITY_RESTRICTED_OPERATION); save_nestlevel = NewGUCNestLevel(); + /* Used for instrumentation and cumulative maintenance statistics. */ + starttime = GetCurrentTimestamp(); + /* measure elapsed time iff autovacuum logging requires it */ if (IsAutoVacuumWorkerProcess() && params->log_min_duration >= 0) { @@ -549,8 +552,6 @@ do_analyze_rel(Relation onerel, VacuumParams *params, } pg_rusage_init(&ru0); - if (params->log_min_duration >= 0) - starttime = GetCurrentTimestamp(); } /* @@ -1206,9 +1207,9 @@ do_analyze_rel(Relation onerel, VacuumParams *params, */ if (!inh) pgstat_report_analyze(onerel, totalrows, totaldeadrows, - (va_cols == NIL)); + (va_cols == NIL), starttime); else if (onerel->rd_rel->relkind == RELKIND_PARTITIONED_TABLE) - pgstat_report_analyze(onerel, 0, 0, (va_cols == NIL)); + pgstat_report_analyze(onerel, 0, 0, (va_cols == NIL), starttime); /* * If this isn't part of VACUUM ANALYZE, let index AMs do cleanup. diff --git a/src/backend/commands/vacuum_ao.c b/src/backend/commands/vacuum_ao.c index f41feed8226..aa07eafacdc 100644 --- a/src/backend/commands/vacuum_ao.c +++ b/src/backend/commands/vacuum_ao.c @@ -321,7 +321,8 @@ ao_vacuum_rel_post_cleanup(Relation onerel, VacuumParams *params, BufferAccessSt pgstat_report_vacuum(RelationGetRelid(onerel), onerel->rd_rel->relisshared, reltuples, - deadtuples); + deadtuples, + vacrelstats->starttime); /* * Remember what is left behind for the vacuum statistics, which @@ -435,6 +436,8 @@ init_vacrelstats() old_context = MemoryContextSwitchTo(TopMemoryContext); vacrelstats = (AOVacuumRelStats *) palloc0(sizeof(AOVacuumRelStats)); + /* Time the run from its first phase in this worker. */ + vacrelstats->starttime = GetCurrentTimestamp(); MemoryContextSwitchTo(old_context); return vacrelstats; @@ -709,6 +712,7 @@ vacuum_appendonly_index(Relation indexRelation, INSTR_TIME_SET_CURRENT(endtime); INSTR_TIME_SUBTRACT(endtime, starttime); + if (!stats) return; diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c index 9d7ba9fa55d..5503096f32c 100644 --- a/src/backend/postmaster/pgstat.c +++ b/src/backend/postmaster/pgstat.c @@ -1589,7 +1589,8 @@ pgstat_report_autovac(Oid dboid) */ void pgstat_report_vacuum(Oid tableoid, bool shared, - PgStat_Counter livetuples, PgStat_Counter deadtuples) + PgStat_Counter livetuples, PgStat_Counter deadtuples, + TimestampTz starttime) { PgStat_MsgVacuum msg; @@ -1601,6 +1602,7 @@ pgstat_report_vacuum(Oid tableoid, bool shared, msg.m_tableoid = tableoid; msg.m_autovacuum = IsAutoVacuumWorkerProcess(); msg.m_vacuumtime = GetCurrentTimestamp(); + msg.m_elapsedtime = Max(msg.m_vacuumtime - starttime, 0); msg.m_live_tuples = livetuples; msg.m_dead_tuples = deadtuples; pgstat_send(&msg, sizeof(msg)); @@ -1618,7 +1620,7 @@ pgstat_report_vacuum(Oid tableoid, bool shared, void pgstat_report_analyze(Relation rel, PgStat_Counter livetuples, PgStat_Counter deadtuples, - bool resetcounter) + bool resetcounter, TimestampTz starttime) { PgStat_MsgAnalyze msg; @@ -1660,6 +1662,7 @@ pgstat_report_analyze(Relation rel, msg.m_autovacuum = IsAutoVacuumWorkerProcess(); msg.m_resetcounter = resetcounter; msg.m_analyzetime = GetCurrentTimestamp(); + msg.m_elapsedtime = Max(msg.m_analyzetime - starttime, 0); msg.m_live_tuples = livetuples; msg.m_dead_tuples = deadtuples; pgstat_send(&msg, sizeof(msg)); @@ -3918,6 +3921,10 @@ pgstat_get_tab_entry(PgStat_StatDBEntry *dbentry, Oid tableoid, bool create) result->autovac_analyze_count = 0; result->frozen_page_marks_cleared = 0; result->visible_page_marks_cleared = 0; + result->total_vacuum_time = 0; + result->total_autovacuum_time = 0; + result->total_analyze_time = 0; + result->total_autoanalyze_time = 0; } return result; @@ -5281,6 +5288,10 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) tabentry->autovac_analyze_count = 0; tabentry->frozen_page_marks_cleared = 0; tabentry->visible_page_marks_cleared = 0; + tabentry->total_vacuum_time = 0; + tabentry->total_autovacuum_time = 0; + tabentry->total_analyze_time = 0; + tabentry->total_autoanalyze_time = 0; } else { @@ -5638,11 +5649,13 @@ pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len) { tabentry->autovac_vacuum_timestamp = msg->m_vacuumtime; tabentry->autovac_vacuum_count++; + tabentry->total_autovacuum_time += msg->m_elapsedtime; } else { tabentry->vacuum_timestamp = msg->m_vacuumtime; tabentry->vacuum_count++; + tabentry->total_vacuum_time += msg->m_elapsedtime; } } @@ -5680,11 +5693,13 @@ pgstat_recv_analyze(PgStat_MsgAnalyze *msg, int len) { tabentry->autovac_analyze_timestamp = msg->m_analyzetime; tabentry->autovac_analyze_count++; + tabentry->total_autoanalyze_time += msg->m_elapsedtime; } else { tabentry->analyze_timestamp = msg->m_analyzetime; tabentry->analyze_count++; + tabentry->total_analyze_time += msg->m_elapsedtime; } } diff --git a/src/include/access/appendonly_compaction.h b/src/include/access/appendonly_compaction.h index fcce48719aa..d8262f759ee 100644 --- a/src/include/access/appendonly_compaction.h +++ b/src/include/access/appendonly_compaction.h @@ -13,6 +13,7 @@ #ifndef APPENDONLY_COMPACTION_H #define APPENDONLY_COMPACTION_H +#include "datatype/timestamp.h" #include "nodes/pg_list.h" #include "access/appendonly_visimap.h" #include "utils/rel.h" @@ -28,6 +29,7 @@ */ typedef struct AOVacuumRelStats { + TimestampTz starttime; /* start of the first vacuum phase in this worker */ int64 nbytes_truncated; /* current # of bytes truncated from segment file */ int64 num_dead_tuples; /* current # of dead tuples */ int num_index_vacuumed; /* current # of indexes been vacuumed */ diff --git a/src/include/pgstat.h b/src/include/pgstat.h index fd9ba381194..b8b1e1fd571 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -419,6 +419,7 @@ typedef struct PgStat_MsgVacuum TimestampTz m_vacuumtime; PgStat_Counter m_live_tuples; PgStat_Counter m_dead_tuples; + PgStat_Counter m_elapsedtime; /* microseconds */ } PgStat_MsgVacuum; @@ -459,6 +460,7 @@ typedef struct PgStat_MsgAnalyze TimestampTz m_analyzetime; PgStat_Counter m_live_tuples; PgStat_Counter m_dead_tuples; + PgStat_Counter m_elapsedtime; /* microseconds */ } PgStat_MsgAnalyze; @@ -755,7 +757,7 @@ typedef union PgStat_Msg * ------------------------------------------------------------ */ -#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA3 +#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA4 /* ---------- * PgStat_StatDBEntry The collector's data per database @@ -846,6 +848,12 @@ typedef struct PgStat_StatTabEntry TimestampTz autovac_analyze_timestamp; /* autovacuum initiated */ PgStat_Counter autovac_analyze_count; + /* Cumulative maintenance times, in microseconds. */ + PgStat_Counter total_vacuum_time; + PgStat_Counter total_autovacuum_time; + PgStat_Counter total_analyze_time; + PgStat_Counter total_autoanalyze_time; + /* VM revisions are fed by ordinary relation statistics. */ PgStat_Counter frozen_page_marks_cleared; PgStat_Counter visible_page_marks_cleared; @@ -1092,7 +1100,8 @@ extern void pgstat_reset_replslot_counter(const char *name); extern void pgstat_report_connect(Oid dboid); extern void pgstat_report_autovac(Oid dboid); extern void pgstat_report_vacuum(Oid tableoid, bool shared, - PgStat_Counter livetuples, PgStat_Counter deadtuples); + PgStat_Counter livetuples, PgStat_Counter deadtuples, + TimestampTz starttime); /* count a page whose all-visible bit is being cleared */ #define pgstat_count_visible_page_marks_cleared(rel) \ @@ -1108,7 +1117,7 @@ extern void pgstat_report_vacuum(Oid tableoid, bool shared, } while (0) extern void pgstat_report_analyze(Relation rel, PgStat_Counter livetuples, PgStat_Counter deadtuples, - bool resetcounter); + bool resetcounter, TimestampTz starttime); extern void pgstat_report_recovery_conflict(int reason); extern void pgstat_report_deadlock(void); From d0775c0a5f87cb8b6ccbc06f53c737d067c3f6a7 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Sat, 18 Jul 2026 21:45:41 +0300 Subject: [PATCH 11/18] Report per-index removed tuples in vacuum instrumentation The per-index autovacuum log line reports page counts but not the number of index entries removed. Add tuples_removed accumulated by the index's bulkdelete passes to that line: index "t_pkey": tuples: 500 removed; pages: 30 in total, ... Port v44-0001 to Cloudberry. On this PG14 base the summary is used by autovacuum; manual VACUUM VERBOSE already reports removed index row versions from lazy_cleanup_one_index(). Preserve that existing output. This uses the access method's existing counter and adds no new statistics collection or dependency on track_vacuum_statistics. Suggested by Bharath Rupireddy (@BRupireddy2). Based-on: https://www.postgresql.org/message-id/attachment/204702/v44-0001-Report-per-index-removed-tuples-in-vacuum-instru.patch --- src/backend/access/heap/vacuumlazy.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 04028f734e0..ad4bb79758c 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -902,8 +902,9 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, continue; appendStringInfo(&buf, - _("index \"%s\": pages: %u in total, %u newly deleted, %u currently deleted, %u reusable\n"), + _("index \"%s\": tuples: %.0f removed; pages: %u in total, %u newly deleted, %u currently deleted, %u reusable\n"), indnames[i], + istat->tuples_removed, istat->num_pages, istat->pages_newly_deleted, istat->pages_deleted, From e02746497f3a1e33add461860484717d945f109e Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Mon, 20 Jul 2026 16:57:07 +0300 Subject: [PATCH 12/18] Track index and database vacuum time and cost-based delay Add cumulative elapsed and delay time for indexes and databases, and delay time for tables. Keep manual VACUUM and autovacuum totals separate. Database totals use table timings, which already include index work. Adapt v44-0002 to Cloudberry's PostgreSQL 14 UDP collector, including AO vacuum. Store times in microseconds and report delay in VACUUM VERBOSE and autovacuum logs. Collection follows track_counts; delay measurement also requires track_cost_delay_timing and excludes ANALYZE calls. SQL access is provided separately through vacuum_stats. Bump the statistics-file format for the new fields; the system catalog is unchanged. Based-on: https://www.postgresql.org/message-id/attachment/204703/v44-0002-Track-vacuum-times-for-indexes-and-databases-and.patch (cherry picked from commit 77a3ffa32d07d8a2947bf7e8dbb8d342ce05549d) Co-authored-by: Andrei Lepikhov Co-authored-by: Andrei Zubkov --- src/backend/access/heap/vacuumlazy.c | 62 ++++++++++++++++--- src/backend/commands/vacuum_ao.c | 5 +- src/backend/postmaster/pgstat.c | 70 +++++++++++++++++++--- src/include/access/appendonly_compaction.h | 1 + src/include/pgstat.h | 19 +++++- 5 files changed, 137 insertions(+), 20 deletions(-) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index ad4bb79758c..ac1260e6c21 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -517,6 +517,9 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, bool aggressive; /* should we scan all unfrozen pages? */ bool scanned_all_unfrozen; /* actually scanned all such pages? */ bool freeze_age_vacuum; /* aggressive due to freeze age? */ + instr_time vacstart; + int64 startdelaytime; + instr_time vacend; char **indnames = NULL; TransactionId xidFullScanLimit; MultiXactId mxactFullScanLimit; @@ -535,6 +538,10 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, /* Used for instrumentation and cumulative maintenance statistics. */ starttime = GetCurrentTimestamp(); + /* measure elapsed and delay time for the vacuum statistics */ + INSTR_TIME_SET_CURRENT(vacstart); + startdelaytime = VacuumDelayTime; + /* measure elapsed time iff autovacuum logging requires it */ if (IsAutoVacuumWorkerProcess() && params->log_min_duration >= 0) { @@ -767,11 +774,13 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, pgstat_report_vacuum(RelationGetRelid(rel), rel->rd_rel->relisshared, Max(new_live_tuples, 0), - vacrel->new_dead_tuples, starttime); + vacrel->new_dead_tuples, starttime, + VacuumDelayTime - startdelaytime); /* assemble the per-vacuum measurements for subsequent reporting */ { PgStat_VacuumStats vacstats; + PgStat_Counter elapsedtime; MemSet(&vacstats, 0, sizeof(vacstats)); vacstats.tuples_deleted = (PgStat_Counter) vacrel->tuples_deleted; @@ -782,9 +791,16 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacstats.pages_all_visible = (PgStat_Counter) vacrel->pages_all_visible; vacstats.freeze_age_vacuum_count = freeze_age_vacuum ? 1 : 0; + INSTR_TIME_SET_CURRENT(vacend); + INSTR_TIME_SUBTRACT(vacend, vacstart); + elapsedtime = (PgStat_Counter) INSTR_TIME_GET_MICROSEC(vacend); + ereport(elevel, (errmsg("table \"%s\": vacuum statistics", vacrel->relname), - errdetail("aggressive scan required by freeze age: %s", + errdetail("elapsed: %.3f ms, cost-based delay: %.3f ms\n" + "aggressive scan required by freeze age: %s", + elapsedtime / 1000.0, + (VacuumDelayTime - startdelaytime) / 1000.0, freeze_age_vacuum ? _("yes") : _("no")))); } @@ -930,6 +946,8 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, (long long) walusage.wal_records, (long long) walusage.wal_fpi, (unsigned long long) walusage.wal_bytes); + appendStringInfo(&buf, _("cost-based delay: %.3f ms\n"), + (VacuumDelayTime - startdelaytime) / 1000.0); appendStringInfo(&buf, _("system usage: %s"), pg_rusage_show(&ru0)); ereport(LOG, @@ -3117,7 +3135,8 @@ lazy_cleanup_all_indexes(LVRelState *vacrel) * is skipped, and keeps it from being counted twice when it is not. */ static IndexBulkDeleteResult -lazy_index_vacstats_start(IndexBulkDeleteResult *istat) +lazy_index_vacstats_start(IndexBulkDeleteResult *istat, + instr_time *starttime, int64 *startdelaytime) { IndexBulkDeleteResult before; @@ -3126,6 +3145,9 @@ lazy_index_vacstats_start(IndexBulkDeleteResult *istat) else MemSet(&before, 0, sizeof(before)); + INSTR_TIME_SET_CURRENT(*starttime); + *startdelaytime = VacuumDelayTime; + return before; } @@ -3139,9 +3161,14 @@ lazy_index_vacstats_start(IndexBulkDeleteResult *istat) */ static void lazy_index_vacstats_finish(Relation indrel, IndexBulkDeleteResult *istat, - IndexBulkDeleteResult *before, bool cleanup) + IndexBulkDeleteResult *before, bool cleanup, + instr_time starttime, int64 startdelaytime) { PgStat_VacuumStats vacstats; + instr_time endtime; + + INSTR_TIME_SET_CURRENT(endtime); + INSTR_TIME_SUBTRACT(endtime, starttime); MemSet(&vacstats, 0, sizeof(vacstats)); if (istat) @@ -3162,6 +3189,11 @@ lazy_index_vacstats_finish(Relation indrel, IndexBulkDeleteResult *istat, (PgStat_Counter) (istat->pages_deleted - istat->pages_free); } + pgstat_report_index_vacuum_time(indrel, + (PgStat_Counter) INSTR_TIME_GET_MICROSEC(endtime), + VacuumDelayTime - startdelaytime, + IsAutoVacuumWorkerProcess()); + } /* @@ -3183,6 +3215,8 @@ lazy_vacuum_one_index(Relation indrel, IndexBulkDeleteResult *istat, PGRUsage ru0; LVSavedErrInfo saved_err_info; IndexBulkDeleteResult istat_before; + instr_time starttime; + int64 startdelaytime; pg_rusage_init(&ru0); @@ -3207,15 +3241,19 @@ lazy_vacuum_one_index(Relation indrel, IndexBulkDeleteResult *istat, InvalidBlockNumber, InvalidOffsetNumber); /* Do bulk deletion */ - istat_before = lazy_index_vacstats_start(istat); + istat_before = lazy_index_vacstats_start(istat, &starttime, + &startdelaytime); istat = index_bulk_delete(&ivinfo, istat, lazy_tid_reaped, (void *) vacrel->dead_tuples); - lazy_index_vacstats_finish(indrel, istat, &istat_before, false); + lazy_index_vacstats_finish(indrel, istat, &istat_before, false, + starttime, startdelaytime); ereport(elevel, (errmsg("scanned index \"%s\" to remove %d row versions", vacrel->indname, vacrel->dead_tuples->num_tuples), - errdetail_internal("%s", pg_rusage_show(&ru0)))); + errdetail("cost-based delay: %.3f ms\n%s", + (VacuumDelayTime - startdelaytime) / 1000.0, + pg_rusage_show(&ru0)))); /* Revert to the previous phase information for error traceback */ restore_vacuum_error_info(vacrel, &saved_err_info); @@ -3242,6 +3280,8 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat, PGRUsage ru0; LVSavedErrInfo saved_err_info; IndexBulkDeleteResult istat_before; + instr_time starttime; + int64 startdelaytime; pg_rusage_init(&ru0); @@ -3266,9 +3306,11 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat, VACUUM_ERRCB_PHASE_INDEX_CLEANUP, InvalidBlockNumber, InvalidOffsetNumber); - istat_before = lazy_index_vacstats_start(istat); + istat_before = lazy_index_vacstats_start(istat, &starttime, + &startdelaytime); istat = index_vacuum_cleanup(&ivinfo, istat); - lazy_index_vacstats_finish(indrel, istat, &istat_before, true); + lazy_index_vacstats_finish(indrel, istat, &istat_before, true, + starttime, startdelaytime); if (istat) { @@ -3280,10 +3322,12 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat, errdetail("%.0f index row versions were removed.\n" "%u index pages were newly deleted.\n" "%u index pages are currently deleted, of which %u are currently reusable.\n" + "cost-based delay: %.3f ms\n" "%s.", (istat)->tuples_removed, (istat)->pages_newly_deleted, (istat)->pages_deleted, (istat)->pages_free, + (VacuumDelayTime - startdelaytime) / 1000.0, pg_rusage_show(&ru0)))); } diff --git a/src/backend/commands/vacuum_ao.c b/src/backend/commands/vacuum_ao.c index aa07eafacdc..1a7f54951fe 100644 --- a/src/backend/commands/vacuum_ao.c +++ b/src/backend/commands/vacuum_ao.c @@ -322,7 +322,9 @@ ao_vacuum_rel_post_cleanup(Relation onerel, VacuumParams *params, BufferAccessSt onerel->rd_rel->relisshared, reltuples, deadtuples, - vacrelstats->starttime); + vacrelstats->starttime, + vacrelstats->delay_time + + (VacuumDelayTime - vacrelstats->phase_start_delay)); /* * Remember what is left behind for the vacuum statistics, which @@ -491,6 +493,7 @@ ao_vacuum_rel(Relation rel, VacuumParams *params, BufferAccessStrategy bstrategy */ INSTR_TIME_SET_CURRENT(phasestart); startdelaytime = VacuumDelayTime; + vacrelstats->phase_start_delay = startdelaytime; if (ao_vacuum_phase == VACOPT_AO_PRE_CLEANUP_PHASE) ao_vacuum_rel_pre_cleanup(rel, params, bstrategy, vacrelstats); diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c index 5503096f32c..1739df7a927 100644 --- a/src/backend/postmaster/pgstat.c +++ b/src/backend/postmaster/pgstat.c @@ -1590,7 +1590,7 @@ pgstat_report_autovac(Oid dboid) void pgstat_report_vacuum(Oid tableoid, bool shared, PgStat_Counter livetuples, PgStat_Counter deadtuples, - TimestampTz starttime) + TimestampTz starttime, PgStat_Counter delaytime) { PgStat_MsgVacuum msg; @@ -1601,6 +1601,8 @@ pgstat_report_vacuum(Oid tableoid, bool shared, msg.m_databaseid = shared ? InvalidOid : MyDatabaseId; msg.m_tableoid = tableoid; msg.m_autovacuum = IsAutoVacuumWorkerProcess(); + msg.m_isindex = false; + msg.m_delaytime = delaytime; msg.m_vacuumtime = GetCurrentTimestamp(); msg.m_elapsedtime = Max(msg.m_vacuumtime - starttime, 0); msg.m_live_tuples = livetuples; @@ -1608,6 +1610,27 @@ pgstat_report_vacuum(Oid tableoid, bool shared, pgstat_send(&msg, sizeof(msg)); } +/* Report an index pass without changing table estimates or vacuum counts. */ +void +pgstat_report_index_vacuum_time(Relation rel, PgStat_Counter elapsedtime, + PgStat_Counter delaytime, bool is_autovacuum) +{ + PgStat_MsgVacuum msg; + + if (pgStatSock == PGINVALID_SOCKET || !pgstat_track_counts) + return; + + MemSet(&msg, 0, sizeof(msg)); + pgstat_setheader(&msg.m_hdr, PGSTAT_MTYPE_VACUUM); + msg.m_databaseid = rel->rd_rel->relisshared ? InvalidOid : MyDatabaseId; + msg.m_tableoid = RelationGetRelid(rel); + msg.m_autovacuum = is_autovacuum; + msg.m_isindex = true; + msg.m_elapsedtime = elapsedtime; + msg.m_delaytime = delaytime; + pgstat_send(&msg, sizeof(msg)); +} + /* -------- * pgstat_report_analyze() - * @@ -3823,9 +3846,14 @@ reset_dbentry_counters(PgStat_StatDBEntry *dbentry) dbentry->n_sessions_abandoned = 0; dbentry->n_sessions_fatal = 0; dbentry->n_sessions_killed = 0; + dbentry->total_vacuum_time = 0; + dbentry->total_autovacuum_time = 0; + dbentry->total_vacuum_delay_time = 0; + dbentry->total_autovacuum_delay_time = 0; dbentry->n_frozen_page_marks_cleared = 0; dbentry->n_visible_page_marks_cleared = 0; + dbentry->stat_reset_timestamp = GetCurrentTimestamp(); dbentry->stats_timestamp = 0; @@ -3919,12 +3947,14 @@ pgstat_get_tab_entry(PgStat_StatDBEntry *dbentry, Oid tableoid, bool create) result->analyze_count = 0; result->autovac_analyze_timestamp = 0; result->autovac_analyze_count = 0; - result->frozen_page_marks_cleared = 0; - result->visible_page_marks_cleared = 0; result->total_vacuum_time = 0; result->total_autovacuum_time = 0; result->total_analyze_time = 0; result->total_autoanalyze_time = 0; + result->total_vacuum_delay_time = 0; + result->total_autovacuum_delay_time = 0; + result->frozen_page_marks_cleared = 0; + result->visible_page_marks_cleared = 0; } return result; @@ -5286,12 +5316,14 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) tabentry->analyze_count = 0; tabentry->autovac_analyze_timestamp = 0; tabentry->autovac_analyze_count = 0; - tabentry->frozen_page_marks_cleared = 0; - tabentry->visible_page_marks_cleared = 0; tabentry->total_vacuum_time = 0; tabentry->total_autovacuum_time = 0; tabentry->total_analyze_time = 0; tabentry->total_autoanalyze_time = 0; + tabentry->total_vacuum_delay_time = 0; + tabentry->total_autovacuum_delay_time = 0; + tabentry->frozen_page_marks_cleared = 0; + tabentry->visible_page_marks_cleared = 0; } else { @@ -5630,6 +5662,32 @@ pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len) tabentry = pgstat_get_tab_entry(dbentry, msg->m_tableoid, true); + if (msg->m_autovacuum) + { + tabentry->total_autovacuum_time += msg->m_elapsedtime; + tabentry->total_autovacuum_delay_time += msg->m_delaytime; + } + else + { + tabentry->total_vacuum_time += msg->m_elapsedtime; + tabentry->total_vacuum_delay_time += msg->m_delaytime; + } + + /* Index passes are already included in the owning table's elapsed time. */ + if (msg->m_isindex) + return; + + if (msg->m_autovacuum) + { + dbentry->total_autovacuum_time += msg->m_elapsedtime; + dbentry->total_autovacuum_delay_time += msg->m_delaytime; + } + else + { + dbentry->total_vacuum_time += msg->m_elapsedtime; + dbentry->total_vacuum_delay_time += msg->m_delaytime; + } + tabentry->n_live_tuples = msg->m_live_tuples; tabentry->n_dead_tuples = msg->m_dead_tuples; @@ -5649,13 +5707,11 @@ pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len) { tabentry->autovac_vacuum_timestamp = msg->m_vacuumtime; tabentry->autovac_vacuum_count++; - tabentry->total_autovacuum_time += msg->m_elapsedtime; } else { tabentry->vacuum_timestamp = msg->m_vacuumtime; tabentry->vacuum_count++; - tabentry->total_vacuum_time += msg->m_elapsedtime; } } diff --git a/src/include/access/appendonly_compaction.h b/src/include/access/appendonly_compaction.h index d8262f759ee..725b56073b0 100644 --- a/src/include/access/appendonly_compaction.h +++ b/src/include/access/appendonly_compaction.h @@ -36,6 +36,7 @@ typedef struct AOVacuumRelStats /* for the vacuum statistics, accumulated over all the phases */ int64 vacuum_time; /* time spent in the phases, in microseconds */ + int64 phase_start_delay; /* counter at the start of the current phase */ int64 delay_time; /* of which the cost-based vacuum delay */ int64 dead_tuples_left; /* tuples the post-cleanup found still hidden */ } AOVacuumRelStats; diff --git a/src/include/pgstat.h b/src/include/pgstat.h index b8b1e1fd571..71bcc441e40 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -416,10 +416,12 @@ typedef struct PgStat_MsgVacuum Oid m_databaseid; Oid m_tableoid; bool m_autovacuum; + bool m_isindex; /* index time does not update tuple/count fields or DB totals */ TimestampTz m_vacuumtime; PgStat_Counter m_live_tuples; PgStat_Counter m_dead_tuples; PgStat_Counter m_elapsedtime; /* microseconds */ + PgStat_Counter m_delaytime; /* microseconds */ } PgStat_MsgVacuum; @@ -442,7 +444,6 @@ typedef struct PgStat_VacuumStats * age thresholds to zero, but not DISABLE_PAGE_SKIPPING alone. */ PgStat_Counter freeze_age_vacuum_count; - } PgStat_VacuumStats; /* ---------- @@ -757,7 +758,7 @@ typedef union PgStat_Msg * ------------------------------------------------------------ */ -#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA4 +#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA5 /* ---------- * PgStat_StatDBEntry The collector's data per database @@ -796,6 +797,13 @@ typedef struct PgStat_StatDBEntry PgStat_Counter n_sessions_fatal; PgStat_Counter n_sessions_killed; + /* Cumulative table vacuum times; index work is already included. */ + PgStat_Counter total_vacuum_time; /* microseconds */ + PgStat_Counter total_autovacuum_time; /* microseconds */ + PgStat_Counter total_vacuum_delay_time; /* microseconds */ + PgStat_Counter total_autovacuum_delay_time; /* microseconds */ + + /* VM revisions are fed by ordinary relation statistics. */ PgStat_Counter n_frozen_page_marks_cleared; PgStat_Counter n_visible_page_marks_cleared; @@ -853,6 +861,8 @@ typedef struct PgStat_StatTabEntry PgStat_Counter total_autovacuum_time; PgStat_Counter total_analyze_time; PgStat_Counter total_autoanalyze_time; + PgStat_Counter total_vacuum_delay_time; + PgStat_Counter total_autovacuum_delay_time; /* VM revisions are fed by ordinary relation statistics. */ PgStat_Counter frozen_page_marks_cleared; @@ -1101,7 +1111,10 @@ extern void pgstat_report_connect(Oid dboid); extern void pgstat_report_autovac(Oid dboid); extern void pgstat_report_vacuum(Oid tableoid, bool shared, PgStat_Counter livetuples, PgStat_Counter deadtuples, - TimestampTz starttime); + TimestampTz starttime, PgStat_Counter delaytime); +extern void pgstat_report_index_vacuum_time(Relation rel, + PgStat_Counter elapsedtime, + PgStat_Counter delaytime, bool is_autovacuum); /* count a page whose all-visible bit is being cleared */ #define pgstat_count_visible_page_marks_cleared(rel) \ From 4ac7782472913b752c9a1138bca1247bbb528747 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Thu, 1 Oct 2026 16:21:14 +0300 Subject: [PATCH 13/18] Count wraparound-failsafe vacuums Port the collection part of v44-0003 to Cloudberry's PG14 statistics collector. Add vacuum_failsafe_count to relation and database statistics. SQL access and TAP coverage follow in the separate vacuum_stats extension commit; built-in functions, system views and the catalog version are unchanged. Failsafe disables cost-based delay and skips index vacuuming and heap truncation to prioritize freezing old XIDs. Count each completed heap vacuum that entered this mode, separately from vacuums made aggressive by the freeze-age threshold. A failsafe event can help identify tables whose regular vacuum schedule or freeze settings need attention. Pass vacrel->failsafe_active through the regular UDP VACUUM report; accumulate the relation and database counters together. Index-pass reports cannot increment these totals. AO reports false because the AO parent has no heap wraparound failsafe; its auxiliary heap relations are accounted for by their own vacuums. Collection follows track_counts. The existing failsafe warning identifies the affected run in the server log. Bump the statistics-file format version for the accumulated counts. Based-on: https://www.postgresql.org/message-id/attachment/204704/v44-0003-Count-wraparound-failsafe-vacuums-in-pg_stat-vie.patch Co-authored-by: Andrei Lepikhov Co-authored-by: Andrei Zubkov --- src/backend/access/heap/vacuumlazy.c | 2 +- src/backend/commands/vacuum_ao.c | 3 ++- src/backend/postmaster/pgstat.c | 13 ++++++++++++- src/include/pgstat.h | 8 ++++++-- 4 files changed, 21 insertions(+), 5 deletions(-) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index ac1260e6c21..b724cf8dbf6 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -775,7 +775,7 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, rel->rd_rel->relisshared, Max(new_live_tuples, 0), vacrel->new_dead_tuples, starttime, - VacuumDelayTime - startdelaytime); + VacuumDelayTime - startdelaytime, vacrel->failsafe_active); /* assemble the per-vacuum measurements for subsequent reporting */ { diff --git a/src/backend/commands/vacuum_ao.c b/src/backend/commands/vacuum_ao.c index 1a7f54951fe..85e52fee89c 100644 --- a/src/backend/commands/vacuum_ao.c +++ b/src/backend/commands/vacuum_ao.c @@ -324,7 +324,8 @@ ao_vacuum_rel_post_cleanup(Relation onerel, VacuumParams *params, BufferAccessSt deadtuples, vacrelstats->starttime, vacrelstats->delay_time + - (VacuumDelayTime - vacrelstats->phase_start_delay)); + (VacuumDelayTime - vacrelstats->phase_start_delay), + false); /* AO itself has no failsafe mode. */ /* * Remember what is left behind for the vacuum statistics, which diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c index 1739df7a927..07d086cc8ad 100644 --- a/src/backend/postmaster/pgstat.c +++ b/src/backend/postmaster/pgstat.c @@ -1590,7 +1590,8 @@ pgstat_report_autovac(Oid dboid) void pgstat_report_vacuum(Oid tableoid, bool shared, PgStat_Counter livetuples, PgStat_Counter deadtuples, - TimestampTz starttime, PgStat_Counter delaytime) + TimestampTz starttime, PgStat_Counter delaytime, + bool failsafe) { PgStat_MsgVacuum msg; @@ -1602,6 +1603,7 @@ pgstat_report_vacuum(Oid tableoid, bool shared, msg.m_tableoid = tableoid; msg.m_autovacuum = IsAutoVacuumWorkerProcess(); msg.m_isindex = false; + msg.m_failsafe = failsafe; msg.m_delaytime = delaytime; msg.m_vacuumtime = GetCurrentTimestamp(); msg.m_elapsedtime = Max(msg.m_vacuumtime - starttime, 0); @@ -3850,6 +3852,7 @@ reset_dbentry_counters(PgStat_StatDBEntry *dbentry) dbentry->total_autovacuum_time = 0; dbentry->total_vacuum_delay_time = 0; dbentry->total_autovacuum_delay_time = 0; + dbentry->vacuum_failsafe_count = 0; dbentry->n_frozen_page_marks_cleared = 0; dbentry->n_visible_page_marks_cleared = 0; @@ -3953,6 +3956,7 @@ pgstat_get_tab_entry(PgStat_StatDBEntry *dbentry, Oid tableoid, bool create) result->total_autoanalyze_time = 0; result->total_vacuum_delay_time = 0; result->total_autovacuum_delay_time = 0; + result->vacuum_failsafe_count = 0; result->frozen_page_marks_cleared = 0; result->visible_page_marks_cleared = 0; } @@ -5322,6 +5326,7 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) tabentry->total_autoanalyze_time = 0; tabentry->total_vacuum_delay_time = 0; tabentry->total_autovacuum_delay_time = 0; + tabentry->vacuum_failsafe_count = 0; tabentry->frozen_page_marks_cleared = 0; tabentry->visible_page_marks_cleared = 0; } @@ -5677,6 +5682,12 @@ pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len) if (msg->m_isindex) return; + if (msg->m_failsafe) + { + tabentry->vacuum_failsafe_count++; + dbentry->vacuum_failsafe_count++; + } + if (msg->m_autovacuum) { dbentry->total_autovacuum_time += msg->m_elapsedtime; diff --git a/src/include/pgstat.h b/src/include/pgstat.h index 71bcc441e40..66c2fda14d8 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -422,6 +422,7 @@ typedef struct PgStat_MsgVacuum PgStat_Counter m_dead_tuples; PgStat_Counter m_elapsedtime; /* microseconds */ PgStat_Counter m_delaytime; /* microseconds */ + bool m_failsafe; /* this completed table vacuum entered failsafe */ } PgStat_MsgVacuum; @@ -758,7 +759,7 @@ typedef union PgStat_Msg * ------------------------------------------------------------ */ -#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA5 +#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA6 /* ---------- * PgStat_StatDBEntry The collector's data per database @@ -802,6 +803,7 @@ typedef struct PgStat_StatDBEntry PgStat_Counter total_autovacuum_time; /* microseconds */ PgStat_Counter total_vacuum_delay_time; /* microseconds */ PgStat_Counter total_autovacuum_delay_time; /* microseconds */ + PgStat_Counter vacuum_failsafe_count; /* VM revisions are fed by ordinary relation statistics. */ @@ -863,6 +865,7 @@ typedef struct PgStat_StatTabEntry PgStat_Counter total_autoanalyze_time; PgStat_Counter total_vacuum_delay_time; PgStat_Counter total_autovacuum_delay_time; + PgStat_Counter vacuum_failsafe_count; /* VM revisions are fed by ordinary relation statistics. */ PgStat_Counter frozen_page_marks_cleared; @@ -1111,7 +1114,8 @@ extern void pgstat_report_connect(Oid dboid); extern void pgstat_report_autovac(Oid dboid); extern void pgstat_report_vacuum(Oid tableoid, bool shared, PgStat_Counter livetuples, PgStat_Counter deadtuples, - TimestampTz starttime, PgStat_Counter delaytime); + TimestampTz starttime, PgStat_Counter delaytime, + bool failsafe); extern void pgstat_report_index_vacuum_time(Relation rel, PgStat_Counter elapsedtime, PgStat_Counter delaytime, bool is_autovacuum); From a1c1b8798d26abaaac79179b043dd17dd3a3f8e2 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Mon, 20 Jul 2026 16:59:26 +0300 Subject: [PATCH 14/18] Count vacuums interrupted by errors Add vacuum_interrupt_count to database statistics for ordinary heap vacuums interrupted by ERROR while the per-relation vacuum error callback is installed. This includes query cancellation. Such a run never reaches the normal end-of-vacuum statistics report. Lower-severity messages carrying vacuum context must not increment the counter. Port v44-0004 to Cloudberry's PG14 collector. Keep the error callback limited to incrementing process-local counters; geterrlevel() distinguishes ERROR from other reports. Include the pending counts in ordinary TABSTAT messages from pgstat_report_stat(), outside the transaction and error path. Send shared-relation errors to the InvalidOid database entry and ensure pending errors are flushed even when there are no table entries to send. Collection follows track_counts and uses ordinary database statistics. Preserve counts across clean restart and clear them with ordinary database reset. Bump the statistics-file format version and adjust TABSTAT payload capacity for its new field. SQL access and TAP coverage follow in the separate vacuum_stats extension commit, without changing the catalog. Document the scope: this does not count VACUUM FULL, errors before the heap callback is installed, or AO parent compaction. Auxiliary heap relations are counted by their own vacuums. The existing ERROR report identifies the failed operation in the log; count it without logging an additional message from the error callback. Original patch reviewed by Vlada Pogozheskaya . Based-on: https://www.postgresql.org/message-id/attachment/204705/v44-0004-Count-vacuums-interrupted-by-errors-in-pg_stat_d.patch Co-authored-by: Andrei Lepikhov Co-authored-by: Andrei Zubkov --- src/backend/access/heap/vacuumlazy.c | 11 +++++++ src/backend/postmaster/pgstat.c | 44 ++++++++++++++++++++++++---- src/backend/utils/error/elog.c | 17 +++++++++++ src/include/pgstat.h | 8 +++-- src/include/utils/elog.h | 1 + 5 files changed, 73 insertions(+), 8 deletions(-) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index b724cf8dbf6..728f34715dc 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -4477,6 +4477,17 @@ vacuum_error_callback(void *arg) { LVRelState *errinfo = arg; + /* + * If an actual ERROR (not a lower-severity report that merely carries + * this vacuum error context) is being raised while we have a relation in + * hand, record at the database level that a vacuum was interrupted. Any + * error here aborts the vacuum, so the exact phase does not matter. We + * are inside the error handler, so this only bumps a counter; the + * statistics are updated at the next pgstat_report_stat(). + */ + if (errinfo->rel != NULL && geterrlevel() == ERROR) + pgstat_count_vacuum_error(errinfo->rel->rd_rel->relisshared); + switch (errinfo->phase) { case VACUUM_ERRCB_PHASE_SCAN_HEAP: diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c index 07d086cc8ad..da3a44dd68e 100644 --- a/src/backend/postmaster/pgstat.c +++ b/src/backend/postmaster/pgstat.c @@ -258,6 +258,10 @@ static PgStat_SubXactStatus *pgStatXactStack = NULL; static int pgStatXactCommit = 0; static int pgStatXactRollback = 0; + +/* Only incremented in error callbacks; sent by pgstat_report_stat(). */ +static PgStat_Counter pgStatVacuumErrors = 0; +static PgStat_Counter pgStatSharedVacuumErrors = 0; PgStat_Counter pgStatBlockReadTime = 0; PgStat_Counter pgStatBlockWriteTime = 0; static PgStat_Counter pgLastSessionReportTime = 0; @@ -893,6 +897,7 @@ pgstat_report_stat(bool disconnect) */ if ((pgStatTabList == NULL || pgStatTabList->tsa_used == 0) && pgStatXactCommit == 0 && pgStatXactRollback == 0 && + pgStatVacuumErrors == 0 && pgStatSharedVacuumErrors == 0 && pgWalUsage.wal_records == prevWalUsage.wal_records && WalStats.m_wal_write == 0 && WalStats.m_wal_sync == 0 && !have_function_stats && !disconnect) @@ -975,13 +980,14 @@ pgstat_report_stat(bool disconnect) /* * Send partial messages. Make sure that any pending xact commit/abort - * and connection stats get counted, even if there are no table stats to - * send. + * counts, connection stats and interrupted vacuums get counted, even if + * there are no table stats to send. */ if (regular_msg.m_nentries > 0 || - pgStatXactCommit > 0 || pgStatXactRollback > 0 || disconnect) + pgStatXactCommit > 0 || pgStatXactRollback > 0 || + pgStatVacuumErrors > 0 || disconnect) pgstat_send_tabstat(®ular_msg, now); - if (shared_msg.m_nentries > 0) + if (shared_msg.m_nentries > 0 || pgStatSharedVacuumErrors > 0) pgstat_send_tabstat(&shared_msg, now); /* Now, send function statistics */ @@ -1008,13 +1014,15 @@ pgstat_send_tabstat(PgStat_MsgTabstat *tsmsg, TimestampTz now) return; /* - * Report and reset accumulated xact commit/rollback and I/O timings - * whenever we send a normal tabstat message + * Report and reset accumulated xact commit/rollback, I/O timings and + * interrupted vacuums whenever we send a normal tabstat message. */ if (OidIsValid(tsmsg->m_databaseid)) { tsmsg->m_xact_commit = pgStatXactCommit; tsmsg->m_xact_rollback = pgStatXactRollback; + tsmsg->m_vacuum_interrupt_count = pgStatVacuumErrors; + pgStatVacuumErrors = 0; tsmsg->m_block_read_time = pgStatBlockReadTime; tsmsg->m_block_write_time = pgStatBlockWriteTime; @@ -1050,6 +1058,8 @@ pgstat_send_tabstat(PgStat_MsgTabstat *tsmsg, TimestampTz now) { tsmsg->m_xact_commit = 0; tsmsg->m_xact_rollback = 0; + tsmsg->m_vacuum_interrupt_count = pgStatSharedVacuumErrors; + pgStatSharedVacuumErrors = 0; tsmsg->m_block_read_time = 0; tsmsg->m_block_write_time = 0; tsmsg->m_session_time = 0; @@ -1612,6 +1622,25 @@ pgstat_report_vacuum(Oid tableoid, bool shared, pgstat_send(&msg, sizeof(msg)); } +/* + * Count a heap vacuum interrupted by ERROR. The caller is an error context + * callback, possibly running while a lock is held. Do not allocate memory, + * acquire locks or send messages here. Like transaction counts, these local + * counters are sent by pgstat_report_stat() outside a transaction. Shared + * relations belong to the InvalidOid database entry. + */ +void +pgstat_count_vacuum_error(bool shared) +{ + if (!pgstat_track_counts) + return; + + if (shared) + pgStatSharedVacuumErrors++; + else + pgStatVacuumErrors++; +} + /* Report an index pass without changing table estimates or vacuum counts. */ void pgstat_report_index_vacuum_time(Relation rel, PgStat_Counter elapsedtime, @@ -3853,6 +3882,8 @@ reset_dbentry_counters(PgStat_StatDBEntry *dbentry) dbentry->total_vacuum_delay_time = 0; dbentry->total_autovacuum_delay_time = 0; dbentry->vacuum_failsafe_count = 0; + dbentry->vacuum_interrupt_count = 0; + dbentry->n_frozen_page_marks_cleared = 0; dbentry->n_visible_page_marks_cleared = 0; @@ -5274,6 +5305,7 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) */ dbentry->n_xact_commit += (PgStat_Counter) (msg->m_xact_commit); dbentry->n_xact_rollback += (PgStat_Counter) (msg->m_xact_rollback); + dbentry->vacuum_interrupt_count += msg->m_vacuum_interrupt_count; dbentry->n_block_read_time += msg->m_block_read_time; dbentry->n_block_write_time += msg->m_block_write_time; diff --git a/src/backend/utils/error/elog.c b/src/backend/utils/error/elog.c index 6c8db2ef5fa..e5566c413ae 100644 --- a/src/backend/utils/error/elog.c +++ b/src/backend/utils/error/elog.c @@ -1647,6 +1647,23 @@ geterrcode(void) return edata->sqlerrcode; } +/* + * geterrlevel --- return the elevel of the error currently being constructed + * + * This is only intended for use in error callback subroutines, where it lets + * a callback tell a genuine error apart from a lower-severity report. + */ +int +geterrlevel(void) +{ + ErrorData *edata = &errordata[errordata_stack_depth]; + + /* we don't bother incrementing recursion_depth */ + CHECK_STACK_DEPTH(); + + return edata->elevel; +} + /* * geterrposition --- return the currently set error position (0 if none) * diff --git a/src/include/pgstat.h b/src/include/pgstat.h index 66c2fda14d8..f02aceea2dc 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -285,7 +285,7 @@ typedef struct PgStat_TableEntry * ---------- */ #define PGSTAT_NUM_TABENTRIES \ - ((PGSTAT_MSG_PAYLOAD - sizeof(Oid) - 3 * sizeof(int) - 5 * sizeof(PgStat_Counter)) \ + ((PGSTAT_MSG_PAYLOAD - sizeof(Oid) - 3 * sizeof(int) - 6 * sizeof(PgStat_Counter)) \ / sizeof(PgStat_TableEntry)) typedef struct PgStat_MsgTabstat @@ -300,6 +300,7 @@ typedef struct PgStat_MsgTabstat PgStat_Counter m_session_time; PgStat_Counter m_active_time; PgStat_Counter m_idle_in_xact_time; + PgStat_Counter m_vacuum_interrupt_count; PgStat_TableEntry m_entry[PGSTAT_NUM_TABENTRIES]; } PgStat_MsgTabstat; @@ -759,7 +760,7 @@ typedef union PgStat_Msg * ------------------------------------------------------------ */ -#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA6 +#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA7 /* ---------- * PgStat_StatDBEntry The collector's data per database @@ -805,6 +806,8 @@ typedef struct PgStat_StatDBEntry PgStat_Counter total_autovacuum_delay_time; /* microseconds */ PgStat_Counter vacuum_failsafe_count; + /* Heap vacuums in this database interrupted by ERROR. */ + PgStat_Counter vacuum_interrupt_count; /* VM revisions are fed by ordinary relation statistics. */ PgStat_Counter n_frozen_page_marks_cleared; @@ -1116,6 +1119,7 @@ extern void pgstat_report_vacuum(Oid tableoid, bool shared, PgStat_Counter livetuples, PgStat_Counter deadtuples, TimestampTz starttime, PgStat_Counter delaytime, bool failsafe); +extern void pgstat_count_vacuum_error(bool shared); extern void pgstat_report_index_vacuum_time(Relation rel, PgStat_Counter elapsedtime, PgStat_Counter delaytime, bool is_autovacuum); diff --git a/src/include/utils/elog.h b/src/include/utils/elog.h index 001bc08a3a5..97f218e907c 100644 --- a/src/include/utils/elog.h +++ b/src/include/utils/elog.h @@ -263,6 +263,7 @@ extern void internalerrquery(const char *query); extern void err_generic_string(int field, const char *str); extern int geterrcode(void); +extern int geterrlevel(void); extern int geterrposition(void); extern int getinternalerrposition(void); From 318586b0a95a491b92688f89f8b540f16c7438af Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Wed, 9 Sep 2026 14:49:54 +0300 Subject: [PATCH 15/18] Report per-relation and per-database vacuum statistics with optional storage Deliver backend measurements through VACSTATS messages and accumulate relation and database counters in the statistics collector. Reuse the existing tuple-removal and heap/index page results; page, VM, timing and failsafe measurements were introduced separately. Keep index work out of database totals where it is already included in table work, and report per-pass index deltas without counting earlier passes again. Retain that work even when a later cleanup is skipped. Add track_vacuum_statistics as a postmaster setting, off by default. Allocate extended vacuum counters with ordinary relation/database entries only when enabled; omit their storage from collector and snapshot hashes when disabled. Keep VM clearings, maintenance times, failsafe and error counts in ordinary entries governed by track_counts. Tag saved statistics so changing the startup setting preserves ordinary counters while disabling tracking discards the extended fields. Configure each node before restart instead of synchronizing the setting through dispatched sessions. Tie the counters and statistics snapshots to their owning relation/database entries and update the statistics-file format. AO reporting follows in a separate commit using the same collector and startup setting. SQL functions, views, documentation and TAP tests follow in the separate vacuum_stats extension commit. The system catalog is unchanged. Cloudberry adaptation of the PostgreSQL extended vacuum statistics collector. Use the PG14 UDP collector instead of v44's report hook and custom statistics kinds. The startup GUC and optional storage are Cloudberry additions. Based-on: https://www.postgresql.org/message-id/flat/cb305107-5935-4c34-9847-6ff0fef89f06%40yandex.ru Co-authored-by: Andrei Lepikhov Co-authored-by: Andrei Zubkov --- src/backend/access/heap/vacuumlazy.c | 10 +- src/backend/postmaster/pgstat.c | 177 ++++++++++++++++-- src/backend/utils/misc/guc.c | 9 + src/backend/utils/misc/postgresql.conf.sample | 1 + src/include/pgstat.h | 42 ++++- src/include/utils/unsync_guc_name.h | 1 + 6 files changed, 217 insertions(+), 23 deletions(-) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 728f34715dc..b8f49a5b204 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -777,7 +777,7 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacrel->new_dead_tuples, starttime, VacuumDelayTime - startdelaytime, vacrel->failsafe_active); - /* assemble the per-vacuum measurements for subsequent reporting */ + /* report the per-vacuum counters as well */ { PgStat_VacuumStats vacstats; PgStat_Counter elapsedtime; @@ -803,6 +803,10 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, (VacuumDelayTime - startdelaytime) / 1000.0, freeze_age_vacuum ? _("yes") : _("no")))); + pgstat_report_vacstats(RelationGetRelid(rel), + rel->rd_rel->relisshared, + false, + &vacstats); } pgstat_progress_end_command(); @@ -3194,6 +3198,10 @@ lazy_index_vacstats_finish(Relation indrel, IndexBulkDeleteResult *istat, VacuumDelayTime - startdelaytime, IsAutoVacuumWorkerProcess()); + pgstat_report_vacstats(RelationGetRelid(indrel), + indrel->rd_rel->relisshared, + true, + &vacstats); } /* diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c index da3a44dd68e..ad32226c252 100644 --- a/src/backend/postmaster/pgstat.c +++ b/src/backend/postmaster/pgstat.c @@ -126,6 +126,7 @@ * ---------- */ bool pgstat_track_counts = false; +bool pgstat_track_vacuum_statistics = false; int pgstat_track_functions = TRACK_FUNC_OFF; bool pgstat_collect_queuelevel = false; @@ -376,6 +377,7 @@ static void pgstat_recv_resetslrucounter(PgStat_MsgResetslrucounter *msg, int le static void pgstat_recv_resetreplslotcounter(PgStat_MsgResetreplslotcounter *msg, int len); static void pgstat_recv_autovac(PgStat_MsgAutovacStart *msg, int len); static void pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len); +static void pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len); static void pgstat_recv_analyze(PgStat_MsgAnalyze *msg, int len); static void pgstat_recv_archiver(PgStat_MsgArchiver *msg, int len); static void pgstat_recv_queuestat(PgStat_MsgQueuestat *msg, int len); /* GPDB */ @@ -1662,6 +1664,32 @@ pgstat_report_index_vacuum_time(Relation rel, PgStat_Counter elapsedtime, pgstat_send(&msg, sizeof(msg)); } +/* --------- + * pgstat_report_vacstats() - + * + * Tell the collector about the counters accumulated while vacuuming a + * relation (a table or an index). isindex tells which one, since only + * the tables count towards the per-database totals. + * --------- + */ +void +pgstat_report_vacstats(Oid tableoid, bool shared, bool isindex, + const PgStat_VacuumStats *stats) +{ + PgStat_MsgVacstats msg; + + if (pgStatSock == PGINVALID_SOCKET || !pgstat_track_counts || + !pgstat_track_vacuum_statistics) + return; + + pgstat_setheader(&msg.m_hdr, PGSTAT_MTYPE_VACSTATS); + msg.m_databaseid = shared ? InvalidOid : MyDatabaseId; + msg.m_tableoid = tableoid; + msg.m_isindex = isindex; + msg.m_stats = *stats; + pgstat_send(&msg, sizeof(msg)); +} + /* -------- * pgstat_report_analyze() - * @@ -2862,6 +2890,25 @@ pgstat_fetch_stat_tabentry(Oid relid) } +/* ---------- + * pgstat_fetch_stat_vacuum_stats() - + * + * Return the vacuum counters available for a relation, or NULL. + * ---------- + */ +PgStat_VacuumStats * +pgstat_fetch_stat_vacuum_stats(Oid relid) +{ + PgStat_StatTabEntry *tabentry; + + if (!pgstat_track_vacuum_statistics) + return NULL; + + tabentry = pgstat_fetch_stat_tabentry(relid); + return tabentry ? &tabentry->vacuum_stats : NULL; +} + + /* ---------- * pgstat_fetch_stat_funcentry() - * @@ -3734,6 +3781,10 @@ PgstatCollectorMain(int argc, char *argv[]) pgstat_recv_vacuum(&msg.msg_vacuum, len); break; + case PGSTAT_MTYPE_VACSTATS: + pgstat_recv_vacstats(&msg.msg_vacstats, len); + break; + case PGSTAT_MTYPE_ANALYZE: pgstat_recv_analyze(&msg.msg_analyze, len); break; @@ -3886,13 +3937,14 @@ reset_dbentry_counters(PgStat_StatDBEntry *dbentry) dbentry->n_frozen_page_marks_cleared = 0; dbentry->n_visible_page_marks_cleared = 0; - + if (pgstat_track_vacuum_statistics) + MemSet(&dbentry->n_vacuum_stats, 0, sizeof(dbentry->n_vacuum_stats)); dbentry->stat_reset_timestamp = GetCurrentTimestamp(); dbentry->stats_timestamp = 0; hash_ctl.keysize = sizeof(Oid); - hash_ctl.entrysize = sizeof(PgStat_StatTabEntry); + hash_ctl.entrysize = PGSTAT_TAB_ENTRY_SIZE; dbentry->tables = hash_create("Per-database table", PGSTAT_TAB_HASH_SIZE, &hash_ctl, @@ -3990,6 +4042,8 @@ pgstat_get_tab_entry(PgStat_StatDBEntry *dbentry, Oid tableoid, bool create) result->vacuum_failsafe_count = 0; result->frozen_page_marks_cleared = 0; result->visible_page_marks_cleared = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&result->vacuum_stats, 0, sizeof(result->vacuum_stats)); } return result; @@ -4096,9 +4150,14 @@ pgstat_write_statsfiles(bool permanent, bool allDbs) * Write out the DB entry. We don't write the tables or functions * pointers, since they're of no use to any other process. */ - fputc('D', fpout); + fputc(pgstat_track_vacuum_statistics ? 'd' : 'D', fpout); rc = fwrite(dbentry, offsetof(PgStat_StatDBEntry, tables), 1, fpout); (void) rc; /* we'll check for error with ferror */ + if (pgstat_track_vacuum_statistics) + { + rc = fwrite(&dbentry->n_vacuum_stats, sizeof(PgStat_VacuumStats), 1, fpout); + (void) rc; + } } /* @@ -4245,8 +4304,8 @@ pgstat_write_db_statsfile(PgStat_StatDBEntry *dbentry, bool permanent) hash_seq_init(&tstat, dbentry->tables); while ((tabentry = (PgStat_StatTabEntry *) hash_seq_search(&tstat)) != NULL) { - fputc('T', fpout); - rc = fwrite(tabentry, sizeof(PgStat_StatTabEntry), 1, fpout); + fputc(pgstat_track_vacuum_statistics ? 't' : 'T', fpout); + rc = fwrite(tabentry, PGSTAT_TAB_ENTRY_SIZE, 1, fpout); (void) rc; /* we'll check for error with ferror */ } @@ -4332,6 +4391,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) HTAB *dbhash; FILE *fpin; int32 format_id; + int record_type; bool found; const char *statfile = permanent ? PGSTAT_STAT_PERMANENT_FILENAME : pgstat_stat_filename; int i; @@ -4349,7 +4409,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) * Create the DB hashtable */ hash_ctl.keysize = sizeof(Oid); - hash_ctl.entrysize = sizeof(PgStat_StatDBEntry); + hash_ctl.entrysize = PGSTAT_DB_ENTRY_SIZE; hash_ctl.hcxt = pgStatLocalContext; dbhash = hash_create("Databases hash", PGSTAT_DB_HASH_SIZE, &hash_ctl, HASH_ELEM | HASH_BLOBS | HASH_CONTEXT); @@ -4479,13 +4539,15 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) */ for (;;) { - switch (fgetc(fpin)) + switch (record_type = fgetc(fpin)) { /* - * 'D' A PgStat_StatDBEntry struct describing a database - * follows. + * 'D' Ordinary database counters follow. + * 'd' The same, followed by a PgStat_VacuumStats block. */ case 'D': + case 'd': + MemSet(&dbbuf, 0, sizeof(dbbuf)); if (fread(&dbbuf, 1, offsetof(PgStat_StatDBEntry, tables), fpin) != offsetof(PgStat_StatDBEntry, tables)) { @@ -4495,6 +4557,15 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) goto done; } + if (record_type == 'd' && + fread(&dbbuf.n_vacuum_stats, 1, sizeof(PgStat_VacuumStats), + fpin) != sizeof(PgStat_VacuumStats)) + { + ereport(pgStatRunningInCollector ? LOG : WARNING, + (errmsg("corrupted statistics file \"%s\"", statfile))); + goto done; + } + /* * Add to the DB hash */ @@ -4510,7 +4581,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) goto done; } - memcpy(dbentry, &dbbuf, sizeof(PgStat_StatDBEntry)); + memcpy(dbentry, &dbbuf, PGSTAT_DB_ENTRY_SIZE); dbentry->tables = NULL; dbentry->functions = NULL; @@ -4535,7 +4606,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) } hash_ctl.keysize = sizeof(Oid); - hash_ctl.entrysize = sizeof(PgStat_StatTabEntry); + hash_ctl.entrysize = PGSTAT_TAB_ENTRY_SIZE; hash_ctl.hcxt = pgStatLocalContext; dbentry->tables = hash_create("Per-database table", PGSTAT_TAB_HASH_SIZE, @@ -4684,6 +4755,8 @@ pgstat_read_db_statsfile(Oid databaseid, HTAB *tabhash, HTAB *funchash, PgStat_StatFuncEntry *funcentry; FILE *fpin; int32 format_id; + int record_type; + size_t tabsize; bool found; char statfile[MAXPGPATH]; @@ -4725,14 +4798,18 @@ pgstat_read_db_statsfile(Oid databaseid, HTAB *tabhash, HTAB *funchash, */ for (;;) { - switch (fgetc(fpin)) + switch (record_type = fgetc(fpin)) { /* - * 'T' A PgStat_StatTabEntry follows. + * 'T' An ordinary table entry follows. + * 't' The entry also includes its vacuum counters. */ case 'T': - if (fread(&tabbuf, 1, sizeof(PgStat_StatTabEntry), - fpin) != sizeof(PgStat_StatTabEntry)) + case 't': + tabsize = record_type == 't' ? sizeof(PgStat_StatTabEntry) : + offsetof(PgStat_StatTabEntry, vacuum_stats); + MemSet(&tabbuf, 0, sizeof(tabbuf)); + if (fread(&tabbuf, 1, tabsize, fpin) != tabsize) { ereport(pgStatRunningInCollector ? LOG : WARNING, (errmsg("corrupted statistics file \"%s\"", @@ -4758,7 +4835,7 @@ pgstat_read_db_statsfile(Oid databaseid, HTAB *tabhash, HTAB *funchash, goto done; } - memcpy(tabentry, &tabbuf, sizeof(tabbuf)); + memcpy(tabentry, &tabbuf, PGSTAT_TAB_ENTRY_SIZE); break; /* @@ -4850,6 +4927,7 @@ pgstat_read_db_statsfile_timestamp(Oid databaseid, bool permanent, PgStat_StatReplSlotEntry myReplSlotStats; FILE *fpin; int32 format_id; + int record_type; const char *statfile = permanent ? PGSTAT_STAT_PERMANENT_FILENAME : pgstat_stat_filename; /* @@ -4933,13 +5011,15 @@ pgstat_read_db_statsfile_timestamp(Oid databaseid, bool permanent, */ for (;;) { - switch (fgetc(fpin)) + switch (record_type = fgetc(fpin)) { /* - * 'D' A PgStat_StatDBEntry struct describing a database - * follows. + * 'D' Ordinary database counters follow. + * 'd' The same, followed by a PgStat_VacuumStats block. */ case 'D': + case 'd': + MemSet(&dbentry, 0, sizeof(dbentry)); if (fread(&dbentry, 1, offsetof(PgStat_StatDBEntry, tables), fpin) != offsetof(PgStat_StatDBEntry, tables)) { @@ -4950,6 +5030,16 @@ pgstat_read_db_statsfile_timestamp(Oid databaseid, bool permanent, return false; } + if (record_type == 'd' && + fread(&dbentry.n_vacuum_stats, 1, sizeof(PgStat_VacuumStats), + fpin) != sizeof(PgStat_VacuumStats)) + { + ereport(pgStatRunningInCollector ? LOG : WARNING, + (errmsg("corrupted statistics file \"%s\"", statfile))); + FreeFile(fpin); + return false; + } + /* * If this is the DB we're looking for, save its timestamp and * we're done. @@ -5361,6 +5451,8 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) tabentry->vacuum_failsafe_count = 0; tabentry->frozen_page_marks_cleared = 0; tabentry->visible_page_marks_cleared = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&tabentry->vacuum_stats, 0, sizeof(tabentry->vacuum_stats)); } else { @@ -5758,6 +5850,53 @@ pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len) } } +/* ---------- + * pgstat_recv_vacstats() - + * + * Process a VACSTATS message: accumulate the vacuum counters into the + * relation's entry and, for a table, into the per-database totals. + * ---------- + */ +static void +pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len) +{ + PgStat_StatDBEntry *dbentry; + PgStat_VacuumStats *vacstats; + + if (!pgstat_track_vacuum_statistics) + return; + + dbentry = pgstat_get_db_entry(msg->m_databaseid, true); + vacstats = &pgstat_get_tab_entry(dbentry, msg->m_tableoid, true)->vacuum_stats; + + vacstats->tuples_deleted += msg->m_stats.tuples_deleted; + vacstats->dead_tuples += msg->m_stats.dead_tuples; + vacstats->pages_deleted += msg->m_stats.pages_deleted; + vacstats->dead_pages += msg->m_stats.dead_pages; + vacstats->pages_frozen += msg->m_stats.pages_frozen; + vacstats->pages_all_visible += msg->m_stats.pages_all_visible; + vacstats->freeze_age_vacuum_count += + msg->m_stats.freeze_age_vacuum_count; + + /* + * The per-database totals describe what vacuum did to the tables. An + * index is vacuumed as a part of its table, and the time it took is + * already accounted for in the table's own report, so adding the index + * counters here would count that work twice. + */ + if (msg->m_isindex) + return; + + dbentry->n_vacuum_stats.tuples_deleted += msg->m_stats.tuples_deleted; + dbentry->n_vacuum_stats.dead_tuples += msg->m_stats.dead_tuples; + dbentry->n_vacuum_stats.pages_deleted += msg->m_stats.pages_deleted; + dbentry->n_vacuum_stats.dead_pages += msg->m_stats.dead_pages; + dbentry->n_vacuum_stats.pages_frozen += msg->m_stats.pages_frozen; + dbentry->n_vacuum_stats.pages_all_visible += msg->m_stats.pages_all_visible; + dbentry->n_vacuum_stats.freeze_age_vacuum_count += + msg->m_stats.freeze_age_vacuum_count; +} + /* ---------- * pgstat_recv_analyze() - * diff --git a/src/backend/utils/misc/guc.c b/src/backend/utils/misc/guc.c index 2f8d498eb20..ef548d23930 100644 --- a/src/backend/utils/misc/guc.c +++ b/src/backend/utils/misc/guc.c @@ -1627,6 +1627,15 @@ static struct config_bool ConfigureNamesBool[] = true, NULL, NULL, NULL }, + { + {"track_vacuum_statistics", PGC_POSTMASTER, STATS_COLLECTOR, + gettext_noop("Collects statistics on what vacuum did and what it cost."), + gettext_noop("The counters are exposed by the vacuum_stats extension.") + }, + &pgstat_track_vacuum_statistics, + false, + NULL, NULL, NULL + }, { {"track_cost_delay_timing", PGC_SUSET, STATS_COLLECTOR, gettext_noop("Collects timing statistics for cost-based vacuum delay."), diff --git a/src/backend/utils/misc/postgresql.conf.sample b/src/backend/utils/misc/postgresql.conf.sample index a5dc8e88745..0487835dd38 100644 --- a/src/backend/utils/misc/postgresql.conf.sample +++ b/src/backend/utils/misc/postgresql.conf.sample @@ -626,6 +626,7 @@ optimizer_analyze_root_partition = on # stats collection on root partitions #track_activities = on #track_activity_query_size = 1024 # (change requires restart) #track_counts = off +#track_vacuum_statistics = off # collect vacuum work and cost (change requires restart) #track_cost_delay_timing = off #track_io_timing = off #track_wal_io_timing = off diff --git a/src/include/pgstat.h b/src/include/pgstat.h index f02aceea2dc..e56cb705d0f 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -85,6 +85,7 @@ typedef enum StatMsgType PGSTAT_MTYPE_REPLSLOT, PGSTAT_MTYPE_CONNECT, PGSTAT_MTYPE_DISCONNECT, + PGSTAT_MTYPE_VACSTATS, } StatMsgType; /* ---------- @@ -448,6 +449,22 @@ typedef struct PgStat_VacuumStats PgStat_Counter freeze_age_vacuum_count; } PgStat_VacuumStats; +/* ---------- + * PgStat_MsgVacstats Sent by the backend or autovacuum daemon + * after vacuuming a heap relation or an index + * to report per-relation vacuum counters. + * ---------- + */ +typedef struct PgStat_MsgVacstats +{ + PgStat_MsgHdr m_hdr; + Oid m_databaseid; + Oid m_tableoid; + bool m_isindex; /* counted apart from the database totals */ + PgStat_VacuumStats m_stats; +} PgStat_MsgVacstats; + + /* ---------- * PgStat_MsgAnalyze Sent by the backend or autovacuum daemon * after ANALYZE @@ -734,6 +751,7 @@ typedef union PgStat_Msg PgStat_MsgResetreplslotcounter msg_resetreplslotcounter; PgStat_MsgAutovacStart msg_autovacuum_start; PgStat_MsgVacuum msg_vacuum; + PgStat_MsgVacstats msg_vacstats; PgStat_MsgAnalyze msg_analyze; PgStat_MsgArchiver msg_archiver; PgStat_MsgQueuestat msg_queuestat; /* GPDB */ @@ -760,7 +778,7 @@ typedef union PgStat_Msg * ------------------------------------------------------------ */ -#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA7 +#define PGSTAT_FILE_FORMAT_ID 0x01A5BCAA /* ---------- * PgStat_StatDBEntry The collector's data per database @@ -817,11 +835,14 @@ typedef struct PgStat_StatDBEntry TimestampTz stats_timestamp; /* time of db stats file update */ /* - * tables and functions must be last in the struct, because we don't write - * the pointers out to the stats file. + * Only the prefix before these pointers is written as ordinary database + * statistics. The optional vacuum counters are serialized separately. */ HTAB *tables; HTAB *functions; + + /* Must be last: storage is omitted when tracking is disabled at startup. */ + PgStat_VacuumStats n_vacuum_stats; } PgStat_StatDBEntry; @@ -873,8 +894,19 @@ typedef struct PgStat_StatTabEntry /* VM revisions are fed by ordinary relation statistics. */ PgStat_Counter frozen_page_marks_cleared; PgStat_Counter visible_page_marks_cleared; + + /* Must be last: storage is omitted when tracking is disabled at startup. */ + PgStat_VacuumStats vacuum_stats; } PgStat_StatTabEntry; +/* The postmaster setting fixes hash entry sizes for the process lifetime. */ +#define PGSTAT_DB_ENTRY_SIZE \ + (pgstat_track_vacuum_statistics ? sizeof(PgStat_StatDBEntry) : \ + offsetof(PgStat_StatDBEntry, n_vacuum_stats)) +#define PGSTAT_TAB_ENTRY_SIZE \ + (pgstat_track_vacuum_statistics ? sizeof(PgStat_StatTabEntry) : \ + offsetof(PgStat_StatTabEntry, vacuum_stats)) + /* ---------- * PgStat_StatQueueEntry The collector's data per resource queue @@ -1043,6 +1075,7 @@ typedef struct PgStat_FunctionCallUsage * ---------- */ extern PGDLLIMPORT bool pgstat_track_counts; +extern PGDLLIMPORT bool pgstat_track_vacuum_statistics; extern PGDLLIMPORT int pgstat_track_functions; extern char *pgstat_stat_directory; extern char *pgstat_stat_tmpname; @@ -1123,6 +1156,8 @@ extern void pgstat_count_vacuum_error(bool shared); extern void pgstat_report_index_vacuum_time(Relation rel, PgStat_Counter elapsedtime, PgStat_Counter delaytime, bool is_autovacuum); +extern void pgstat_report_vacstats(Oid tableoid, bool shared, bool isindex, + const PgStat_VacuumStats *stats); /* count a page whose all-visible bit is being cleared */ #define pgstat_count_visible_page_marks_cleared(rel) \ @@ -1346,6 +1381,7 @@ extern void pgstat_combine_from_qe(struct CdbDispatchResults *results, /* GPDB * */ extern PgStat_StatDBEntry *pgstat_fetch_stat_dbentry(Oid dbid); extern PgStat_StatTabEntry *pgstat_fetch_stat_tabentry(Oid relid); +extern PgStat_VacuumStats *pgstat_fetch_stat_vacuum_stats(Oid relid); extern PgStat_StatQueueEntry *pgstat_fetch_stat_queueentry(Oid queueid); /* GPDB */ extern PgBackendStatus *pgstat_fetch_stat_beentry(int beid); diff --git a/src/include/utils/unsync_guc_name.h b/src/include/utils/unsync_guc_name.h index 4bfb7d5d06e..9b98358a478 100644 --- a/src/include/utils/unsync_guc_name.h +++ b/src/include/utils/unsync_guc_name.h @@ -603,6 +603,7 @@ "track_counts", "track_functions", "track_io_timing", + "track_vacuum_statistics", "transaction_deferrable", "transaction_isolation", "transaction_read_only", From a6f12a8f5d2ee1456822219a68eba17e777738d8 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Mon, 21 Sep 2026 01:40:05 +0300 Subject: [PATCH 16/18] Report append-optimized vacuum statistics Report existing compaction tuple/byte counts and remaining dead tuples. Connect AO index-pass elapsed and cost-delay time to the ordinary counters, including cleanup without obsolete segments or AM page results. Include obsolete index entries for relocated rows and exclude index work from database totals. AO itself has no heap VM or wraparound failsafe. Record exact bytes truncated from heap and AO table files in relation and database statistics. Record the AO segment metadata count after the latest vacuum as total_file_segs, replacing this state on each report. Update the statistics-file format for the added fields. Include truncated bytes and segment counts in AO VERBOSE output. Build and send extended AO table/index reports only when the startup track_vacuum_statistics setting is enabled. Keep ordinary index timing outside that guard so it continues to follow track_counts, including cleanup-only scans. VERBOSE measurements remain available with tracking off. SQL access, documentation and tests for AO row and column tables follow in the separate vacuum_stats extension commit. The system catalog is unchanged. --- src/backend/access/heap/vacuumlazy.c | 1 + src/backend/commands/vacuum_ao.c | 84 ++++++++++++++++++++++ src/backend/postmaster/pgstat.c | 4 ++ src/include/access/appendonly_compaction.h | 1 + src/include/pgstat.h | 4 +- 5 files changed, 93 insertions(+), 1 deletion(-) diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index b8f49a5b204..97c03036656 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -786,6 +786,7 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacstats.tuples_deleted = (PgStat_Counter) vacrel->tuples_deleted; vacstats.dead_tuples = (PgStat_Counter) vacrel->new_dead_tuples; vacstats.pages_deleted = (PgStat_Counter) vacrel->pages_removed; + vacstats.bytes_removed = (PgStat_Counter) vacrel->pages_removed * BLCKSZ; vacstats.dead_pages = (PgStat_Counter) vacrel->dead_pages; vacstats.pages_frozen = (PgStat_Counter) vacrel->pages_frozen; vacstats.pages_all_visible = (PgStat_Counter) vacrel->pages_all_visible; diff --git a/src/backend/commands/vacuum_ao.c b/src/backend/commands/vacuum_ao.c index 85e52fee89c..72435bd03ba 100644 --- a/src/backend/commands/vacuum_ao.c +++ b/src/backend/commands/vacuum_ao.c @@ -332,6 +332,7 @@ ao_vacuum_rel_post_cleanup(Relation onerel, VacuumParams *params, BufferAccessSt * ao_vacuum_rel() reports once this last phase is over. */ vacrelstats->dead_tuples_left = (int64) deadtuples; + vacrelstats->total_file_segs = total_file_segs; SIMPLE_FAULT_INJECTOR("vacuum_ao_post_cleanup_end"); } @@ -446,6 +447,69 @@ init_vacrelstats() return vacrelstats; } +/* + * Report what the vacuuming of an append-optimized relation did to the + * statistics collector, the way lazy vacuum does for a heap relation. + * + * The counters that describe heap pages have no counterpart here: an AO + * relation has no heap visibility map and nothing to freeze, so + * pages_frozen and pages_all_visible stay zero. Nor is there a relfrozenxid + * of its own to reach the freeze table age -- it is always invalid, the + * auxiliary heap relations being vacuumed and frozen on their own -- so the + * freeze_age_vacuum_count stays zero as well. What the heap reports as + * truncated pages is the space compaction freed, measured in blocks of the + * segment files it dropped or truncated. + * + * The indexes report themselves, from vacuum_appendonly_index() and scan_index(). + */ +static void +ao_report_vacuum_stats(Relation aorel, AOVacuumRelStats *vacrelstats) +{ + PgStat_VacuumStats vacstats; + + Assert(pgstat_track_vacuum_statistics); + + MemSet(&vacstats, 0, sizeof(vacstats)); + vacstats.tuples_deleted = (PgStat_Counter) vacrelstats->num_dead_tuples; + vacstats.dead_tuples = (PgStat_Counter) vacrelstats->dead_tuples_left; + /* Keep the conversion 64-bit: aggregate AO files can exceed BlockNumber. */ + vacstats.bytes_removed = vacrelstats->nbytes_truncated; + vacstats.pages_deleted = vacrelstats->nbytes_truncated / BLCKSZ + + (vacrelstats->nbytes_truncated % BLCKSZ != 0); + vacstats.total_file_segs = vacrelstats->total_file_segs; + + pgstat_report_vacstats(RelationGetRelid(aorel), + aorel->rd_rel->relisshared, + false, + &vacstats); +} + +/* Report index work, including cleanup runs with no obsolete AO segments. */ +static void +ao_report_index_vacuum_stats(Relation indexRelation, + IndexBulkDeleteResult *stats) +{ + PgStat_VacuumStats vacstats; + + Assert(pgstat_track_vacuum_statistics); + + MemSet(&vacstats, 0, sizeof(vacstats)); + if (stats) + { + vacstats.tuples_deleted = (PgStat_Counter) stats->tuples_removed; + vacstats.pages_deleted = (PgStat_Counter) stats->pages_newly_deleted; + /* deleted pages that are not yet reusable still hold dead entries */ + if (stats->pages_deleted > stats->pages_free) + vacstats.dead_pages = + (PgStat_Counter) (stats->pages_deleted - stats->pages_free); + } + + pgstat_report_vacstats(RelationGetRelid(indexRelation), + indexRelation->rd_rel->relisshared, + true, + &vacstats); +} + /* * ao_vacuum_rel() * @@ -522,11 +586,16 @@ ao_vacuum_rel(Relation rel, VacuumParams *params, BufferAccessStrategy bstrategy (errmsg("append-optimized table \"%s\": vacuum statistics", RelationGetRelationName(rel)), errdetail("%lld dead tuples remain.\n" + "%lld bytes truncated; %lld file segments remain.\n" "elapsed: %.3f ms, cost-based delay: %.3f ms", (long long) vacrelstats->dead_tuples_left, + (long long) vacrelstats->nbytes_truncated, + (long long) vacrelstats->total_file_segs, vacrelstats->vacuum_time / 1000.0, vacrelstats->delay_time / 1000.0))); + if (pgstat_track_vacuum_statistics) + ao_report_vacuum_stats(rel, vacrelstats); pgstat_progress_end_command(); cleanup_vacrelstats(&vacrelstats); } @@ -716,6 +785,13 @@ vacuum_appendonly_index(Relation indexRelation, INSTR_TIME_SET_CURRENT(endtime); INSTR_TIME_SUBTRACT(endtime, starttime); + /* Ordinary timing follows track_counts, independently of work counters. */ + pgstat_report_index_vacuum_time(indexRelation, + (PgStat_Counter) INSTR_TIME_GET_MICROSEC(endtime), + VacuumDelayTime - startdelaytime, + IsAutoVacuumWorkerProcess()); + if (pgstat_track_vacuum_statistics) + ao_report_index_vacuum_stats(indexRelation, stats); if (!stats) return; @@ -890,6 +966,14 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac INSTR_TIME_SET_CURRENT(endtime); INSTR_TIME_SUBTRACT(endtime, starttime); + /* Ordinary timing follows track_counts, independently of work counters. */ + pgstat_report_index_vacuum_time(indrel, + (PgStat_Counter) INSTR_TIME_GET_MICROSEC(endtime), + VacuumDelayTime - startdelaytime, + IsAutoVacuumWorkerProcess()); + if (pgstat_track_vacuum_statistics) + ao_report_index_vacuum_stats(indrel, stats); + if (!stats) return; diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c index ad32226c252..7b2223a1f59 100644 --- a/src/backend/postmaster/pgstat.c +++ b/src/backend/postmaster/pgstat.c @@ -5872,6 +5872,9 @@ pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len) vacstats->tuples_deleted += msg->m_stats.tuples_deleted; vacstats->dead_tuples += msg->m_stats.dead_tuples; vacstats->pages_deleted += msg->m_stats.pages_deleted; + vacstats->bytes_removed += msg->m_stats.bytes_removed; + /* Relation state is replaced, never added to database totals. */ + vacstats->total_file_segs = msg->m_stats.total_file_segs; vacstats->dead_pages += msg->m_stats.dead_pages; vacstats->pages_frozen += msg->m_stats.pages_frozen; vacstats->pages_all_visible += msg->m_stats.pages_all_visible; @@ -5890,6 +5893,7 @@ pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len) dbentry->n_vacuum_stats.tuples_deleted += msg->m_stats.tuples_deleted; dbentry->n_vacuum_stats.dead_tuples += msg->m_stats.dead_tuples; dbentry->n_vacuum_stats.pages_deleted += msg->m_stats.pages_deleted; + dbentry->n_vacuum_stats.bytes_removed += msg->m_stats.bytes_removed; dbentry->n_vacuum_stats.dead_pages += msg->m_stats.dead_pages; dbentry->n_vacuum_stats.pages_frozen += msg->m_stats.pages_frozen; dbentry->n_vacuum_stats.pages_all_visible += msg->m_stats.pages_all_visible; diff --git a/src/include/access/appendonly_compaction.h b/src/include/access/appendonly_compaction.h index 725b56073b0..59a552adfee 100644 --- a/src/include/access/appendonly_compaction.h +++ b/src/include/access/appendonly_compaction.h @@ -39,6 +39,7 @@ typedef struct AOVacuumRelStats int64 phase_start_delay; /* counter at the start of the current phase */ int64 delay_time; /* of which the cost-based vacuum delay */ int64 dead_tuples_left; /* tuples the post-cleanup found still hidden */ + int64 total_file_segs; /* segment metadata entries after post-cleanup */ } AOVacuumRelStats; extern Bitmapset *AppendOptimizedCollectDeadSegments(Relation aorel); diff --git a/src/include/pgstat.h b/src/include/pgstat.h index e56cb705d0f..6558eeae6ce 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -437,7 +437,9 @@ typedef struct PgStat_VacuumStats PgStat_Counter tuples_deleted; /* tuples removed by vacuum */ PgStat_Counter dead_tuples; /* dead tuples left unremoved */ PgStat_Counter pages_deleted; /* pages removed/deleted by vacuum */ + PgStat_Counter bytes_removed; /* bytes physically truncated from table files */ PgStat_Counter dead_pages; /* pages with unremoved dead tuples */ + PgStat_Counter total_file_segs; /* latest AO segment count, not cumulative */ PgStat_Counter pages_frozen; /* pages where vacuum froze tuples */ PgStat_Counter pages_all_visible; /* pages marked all-visible by vacuum */ @@ -778,7 +780,7 @@ typedef union PgStat_Msg * ------------------------------------------------------------ */ -#define PGSTAT_FILE_FORMAT_ID 0x01A5BCAA +#define PGSTAT_FILE_FORMAT_ID 0x01A5BCAD /* ---------- * PgStat_StatDBEntry The collector's data per database From 628ab63f5867a1a306fbb8b2fccb05ea00ed2c40 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Mon, 21 Sep 2026 01:40:05 +0300 Subject: [PATCH 17/18] Expose vacuum statistics through the vacuum_stats extension Add SQL getters and local pg_stat_vacuum_tables, pg_stat_vacuum_indexes and pg_stat_vacuum_database views for the previously introduced statistics. Read VM clearings, maintenance times, failsafe and interruption counts from ordinary collector entries, and work counters from the optional vacuum statistics storage. Keep all getters in the extension library and schema; no built-in OIDs, system views or catalog-version changes are required. Expose the timing backport's separate table ANALYZE and autoanalyze totals alongside manual/autovacuum elapsed and delay times. VACUUM delay totals exclude calls classified as ANALYZE. ANALYZE elapsed time has its own counters; there is no separate ANALYZE delay counter. Test standalone ANALYZE and VACUUM ANALYZE with cost-based throttling enabled. Include datid zero for shared-relation statistics. Add coordinator and segment wrappers and gp_stat_vacuum_* views with gp_segment_id. Suppress the segment branch in utility mode so local rows appear exactly once. Resolve extension getters explicitly in the extension's installation schema. Document counter meanings and their use in evaluating vacuum effectiveness. Add TAP coverage for heap and index work, AO row/column compaction, VM transitions, manual and automatic timing, failsafe, canceled vacuums, startup allocation and persistence. Adapt the v44 scenarios to the PG14 UDP collector and wait for observable reports instead of using fixed sleeps. Cloudberry adaptation of the PostgreSQL extended vacuum statistics SQL interface. Read counters already collected by the backend, including timing, failsafe and VM statistics, through the extension without changing Cloudberry's system catalog. Based-on: https://www.postgresql.org/message-id/flat/cb305107-5935-4c34-9847-6ff0fef89f06%40yandex.ru --- contrib/Makefile | 1 + contrib/vacuum_stats/.gitignore | 4 + contrib/vacuum_stats/Makefile | 29 + contrib/vacuum_stats/README.md | 134 ++++ .../vacuum_stats/t/001_vacuum_statistics.pl | 693 ++++++++++++++++++ .../vacuum_stats/t/002_index_vacuum_time.pl | 252 +++++++ contrib/vacuum_stats/t/003_vacuum_failsafe.pl | 113 +++ .../t/004_visibility_map_stats.pl | 216 ++++++ .../vacuum_stats/t/005_vacuum_interrupts.pl | 133 ++++ contrib/vacuum_stats/vacuum_stats--1.0.sql | 558 ++++++++++++++ contrib/vacuum_stats/vacuum_stats.c | 160 ++++ contrib/vacuum_stats/vacuum_stats.control | 5 + 12 files changed, 2298 insertions(+) create mode 100644 contrib/vacuum_stats/.gitignore create mode 100644 contrib/vacuum_stats/Makefile create mode 100644 contrib/vacuum_stats/README.md create mode 100644 contrib/vacuum_stats/t/001_vacuum_statistics.pl create mode 100644 contrib/vacuum_stats/t/002_index_vacuum_time.pl create mode 100644 contrib/vacuum_stats/t/003_vacuum_failsafe.pl create mode 100644 contrib/vacuum_stats/t/004_visibility_map_stats.pl create mode 100644 contrib/vacuum_stats/t/005_vacuum_interrupts.pl create mode 100644 contrib/vacuum_stats/vacuum_stats--1.0.sql create mode 100644 contrib/vacuum_stats/vacuum_stats.c create mode 100644 contrib/vacuum_stats/vacuum_stats.control diff --git a/contrib/Makefile b/contrib/Makefile index 5ea76366363..5d8e38819d6 100644 --- a/contrib/Makefile +++ b/contrib/Makefile @@ -54,6 +54,7 @@ SUBDIRS = \ tsm_system_rows \ tsm_system_time \ unaccent \ + vacuum_stats \ vacuumlo # Cloudberry-specific additions (to ease merge pain). diff --git a/contrib/vacuum_stats/.gitignore b/contrib/vacuum_stats/.gitignore new file mode 100644 index 00000000000..5dcb3ff9723 --- /dev/null +++ b/contrib/vacuum_stats/.gitignore @@ -0,0 +1,4 @@ +# Generated subdirectories +/log/ +/results/ +/tmp_check/ diff --git a/contrib/vacuum_stats/Makefile b/contrib/vacuum_stats/Makefile new file mode 100644 index 00000000000..aee5f608665 --- /dev/null +++ b/contrib/vacuum_stats/Makefile @@ -0,0 +1,29 @@ +# contrib/vacuum_stats/Makefile + +MODULE_big = vacuum_stats +OBJS = vacuum_stats.o + +EXTENSION = vacuum_stats +DATA = vacuum_stats--1.0.sql +PGFILEDESC = "vacuum_stats - per-relation and per-database vacuum statistics" + +TAP_TESTS = 1 + +ifdef USE_PGXS +PG_CONFIG = pg_config +PGXS := $(shell $(PG_CONFIG) --pgxs) +include $(PGXS) +else +subdir = contrib/vacuum_stats +top_builddir = ../.. +include $(top_builddir)/src/Makefile.global +include $(top_srcdir)/contrib/contrib-global.mk +endif + +check-tap: + $(prove_check) + +installcheck-tap: + $(prove_installcheck) + +.PHONY: check-tap installcheck-tap diff --git a/contrib/vacuum_stats/README.md b/contrib/vacuum_stats/README.md new file mode 100644 index 00000000000..a0cf2982e80 --- /dev/null +++ b/contrib/vacuum_stats/README.md @@ -0,0 +1,134 @@ +# vacuum_stats + +`vacuum_stats` describes the work done by VACUUM for tables, indexes and +whole databases: how many tuples it removes, what remains to be cleaned, +how it changes page visibility, and how much time it spends. These counters +help evaluate the results and cost of vacuuming over time, alongside the +existing vacuum counts and timestamps. + +Collection of extended work counters is enabled with `track_vacuum_statistics = on` in the server +configuration and requires a restart. Set it consistently on the coordinator +and segments. When enabled, the extended vacuum counters and database totals +are stored with ordinary relation and database statistics. When disabled, their storage +is omitted and those extended fields return zero. Restarting with tracking +disabled discards the extended counters while preserving ordinary statistics, +including vacuum times, VM clearings, failsafe and interruption counts. + +The `visible_page_marks_cleared` and `frozen_page_marks_cleared` counters describe +visibility-map changes caused by data modifications. They follow `track_counts` +and are collected and retained independently of `track_vacuum_statistics`. + +Install with `CREATE EXTENSION vacuum_stats`. All new SQL functions and +views belong to the extension's schema. Existing system views and built-in +function OIDs remain unchanged. VM clearings, maintenance times, failsafe and +interruption counts follow `track_counts` independently of +`track_vacuum_statistics`; the extension reads their ordinary collector entries. + +Cost-based delay timing additionally requires `track_cost_delay_timing = on`. +It is disabled by default and can be changed for a session without restarting. +ANALYZE elapsed times are recorded separately from VACUUM times. ANALYZE +sampling delays do not contribute to the VACUUM delay counters. + +The counters accumulate until reset, except `total_file_segs`, which records +the state observed by the last completed AO VACUUM. To examine a particular period, compare +two readings without an intervening reset or change in tracking configuration. + +## What the counters measure + +| Counter | Meaning | +|---------|---------| +| `tuples_deleted` | Table tuples or index entries removed by VACUUM. | +| `dead_tuples` | Heap tuples found dead but not yet removable; for AO tables, hidden tuples remaining after vacuum. | +| `pages_deleted` | Pages truncated from a heap, newly deleted index pages, or space freed from AO segment files expressed in blocks. | +| `bytes_removed` | Bytes physically truncated from heap or AO table files, including AO tails left by aborted inserts. Index page reuse does not increase this counter. | +| `dead_pages` | Heap pages containing unremovable dead tuples; for indexes, deleted pages not yet available for reuse. | +| `pages_frozen` | Heap pages on which VACUUM froze at least one tuple. | +| `pages_all_visible` | Heap pages VACUUM marked all-visible. | +| `visible_page_marks_cleared` | Clearings of the all-visible flag, usually caused by data changes. | +| `frozen_page_marks_cleared` | Clearings of the all-frozen flag, usually caused by data changes. | +| `freeze_age_vacuum_count` | Heap vacuum runs made aggressive by transaction or multixact freeze age. This includes `VACUUM FREEZE`; forcing page scanning alone does not increment it. | +| `vacuum_failsafe_count` | Completed heap vacuum runs that entered failsafe mode to avoid transaction or multixact wraparound. Aggressive scanning alone does not increment it. | +| `total_vacuum_time`, `total_autovacuum_time` | Elapsed time in milliseconds, separately for manual VACUUM and autovacuum. | +| `total_vacuum_delay_time`, `total_autovacuum_delay_time` | Cost-based delays in milliseconds, included in the corresponding elapsed time. | +| `total_analyze_time`, `total_autoanalyze_time` | Table ANALYZE time in milliseconds, separately for manual and automatic runs. | +| `vacuum_interrupt_count` | Database count of heap vacuums interrupted by an ERROR, including cancellation, while the vacuum error callback is installed. | + +`total_file_segs` in the table views records the number of AO segment metadata +entries observed after the last VACUUM, including empty and awaiting-drop +segments. For AO column tables it counts logical segments, not each column +file. A subsequent VACUUM replaces this value instead of adding to it. It is +zero before the first report, after reset, and for heap tables. Use it with +`bytes_removed` and remaining hidden tuples to understand compaction results. + +The other fields are cumulative counters, not a snapshot of the table's current +contents. In particular, successive runs can count the same unremovable +tuple or page again. Visibility flags can also be set and cleared repeatedly. +A counter difference measures work or observations during the interval, +not necessarily a number of distinct tuples or pages. + +## How to use them + +- **Find expensive relations.** Compare increases in `total_vacuum_time` and + `total_autovacuum_time` between tables and indexes over the same interval. Relate that time to tuples + removed and pages reclaimed to see where maintenance time is spent. + Low tuple removal alone does not imply wasted work: vacuum also freezes + tuples and maintains visibility information. +- **Find work that cannot finish.** Repeated increases in heap `dead_tuples` + and `dead_pages` show that vacuum keeps encountering data it cannot remove. + Check for old snapshots or long-running transactions before increasing + vacuum frequency. +- **Separate throttling from other costs.** Compare `total_vacuum_delay_time` with + `total_vacuum_time`, and the corresponding autovacuum counters. A large share + spent in cost-based delays helps explain a long run. The remaining time includes execution and other waits; it is + not a measurement of CPU time. +- **Understand visibility and freezing work.** Compare `pages_all_visible` + with `visible_page_marks_cleared` to see how often data changes undo visibility + work. Use `pages_frozen`, `frozen_page_marks_cleared` and + `freeze_age_vacuum_count` to understand freezing activity and aggressive + scans. Zero frozen pages can be normal when no tuples need freezing. +- **Compare segments.** Differences in work and time for the same relation + can help identify uneven data distribution or different execution costs. + Compare both quantities: a slower segment is not necessarily processing + more data. Summed segment time represents accumulated work, not the + wall-clock duration of a distributed VACUUM. + +## Tables, indexes and AO + +A table's vacuum time includes its index maintenance. Database totals +include table work without adding the index counters again. Use the index +figures to understand that part of the cost, rather than adding them to +table totals. Deleted index pages can become reusable within the index; +they do not necessarily represent space returned to the filesystem. + +For AO row and column tables, `tuples_deleted` measures rows discarded by +compaction, and `dead_tuples` counts hidden rows left afterwards. Hidden +rows can remain when compaction is disabled or a segment file is not +eligible for compaction. AO index cleanup can remove entries for relocated +live rows as well as deleted rows, so its tuple count can exceed the table's. +AO `pages_deleted` expresses truncated bytes in blocks, rounded up; +`bytes_removed` preserves the exact byte count. Freed space can include live +rows moved to new segment files, so this is not the net reduction in table size. +AO elapsed time covers the interval from the first phase seen by the +reporting worker to final cleanup, including gaps between those phases. + +Heap page visibility, freezing and failsafe counters do not apply to the AO table +itself. Its auxiliary heap relations have their own statistics. Indexes +have tuple-removal, page-deletion and timing counters, but no heap visibility +or freezing work. + +The `pg_stat_vacuum_tables`, `pg_stat_vacuum_indexes` and +`pg_stat_vacuum_database` views expose local statistics. Their +`gp_stat_vacuum_*` counterparts include the coordinator and segments, +identified by `gp_segment_id`. In utility mode there is no dispatch: these +views return the connected node's local statistics once, with its segment ID. + +The database views include `datid = 0`, with a null `datname`, for shared +relations. `vacuum_interrupt_count` does not include VACUUM FULL, failures +before the heap callback is installed, or AO parent compaction. Auxiliary +heap vacuums are counted independently. An interrupted run does not report +its usual completion counters, so check errors when successful-run totals +alone do not explain maintenance activity. + +The on-disk statistics format changes, so older saved statistics are discarded +on first start. The system catalog version is unchanged; these statistics do +not require a new cluster or replacement of system views. diff --git a/contrib/vacuum_stats/t/001_vacuum_statistics.pl b/contrib/vacuum_stats/t/001_vacuum_statistics.pl new file mode 100644 index 00000000000..bf25446f135 --- /dev/null +++ b/contrib/vacuum_stats/t/001_vacuum_statistics.pl @@ -0,0 +1,693 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Test vacuum counter semantics, controls and lifecycle in one TAP suite. +# Adapt the repeatable-read and visibility-map scenarios from v44: +# https://www.postgresql.org/message-id/flat/cb305107-5935-4c34-9847-6ff0fef89f06%40yandex.ru#8b78d4ab399e37afc5eb2cee8a54d2f6 +# This branch has a UDP collector and the older pg_stat_vacuum_* API: +# dead_tuples corresponds to recently_dead_tuples, and the page counters +# expose freezing and VM transitions. Polling uses a fresh connection, so +# it does not reuse a cached statistics snapshot. Core VM clearing semantics +# are covered by 004_visibility_map_stats.pl; here check vacuum +# reports, the extension views, controls and lifecycle. +# PostgresNode runs in utility/maintenance mode; use the local views. +# Read ordinary counters from pg_stat_all_tables_internal: the public +# pg_stat_all_tables gathers segment data and has no user-table rows here. + +use strict; +use warnings; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('vacuum_stats'); +$node->init; +$node->append_conf('postgresql.conf', q{ +autovacuum = off +track_counts = on +vacuum_cost_delay = 0 +}); +$node->start; +$node->safe_psql('postgres', q{ +CREATE EXTENSION vacuum_stats; +CREATE TABLE vstat_barrier (id int) WITH (autovacuum_enabled = off); +}); + +my @counter_names = qw(tuples_deleted dead_tuples pages_deleted bytes_removed dead_pages + pages_frozen pages_all_visible frozen_page_marks_cleared visible_page_marks_cleared + freeze_age_vacuum_count vacuum_failsafe_count total_vacuum_time total_autovacuum_time + total_vacuum_delay_time total_autovacuum_delay_time); +my $all_counters = join(', ', @counter_names); +my $all_zero = join(' AND ', map { "$_ = 0" } @counter_names); +my $vacuum_zero = join(' AND ', map { "$_ = 0" } grep { !/_page_marks_cleared$|^total_.*time$|^vacuum_failsafe_count$/ } @counter_names); +my $vm_columns = 'frozen_page_marks_cleared, visible_page_marks_cleared'; + +sub wait_for_stats +{ + my ($sql, $description, $expected) = @_; + $expected = 't' unless defined $expected; + $node->poll_query_until('postgres', $sql, $expected) + or BAIL_OUT("timed out waiting for $description: $sql"); +} + +# VACUUM sends vacuum_count before its extended counters. Instead of +# using that earlier report, send ANALYZE from the same backend AFTER the +# command and wait for its report. ANALYZE adds no vacuum counters, so it +# also works for database totals, disabled tracking, resets and VACUUM FULL. +# DML counters are buffered separately; wait for the affected table's DML +# report explicitly when checking VM clearing counters. +sub run_and_wait +{ + my ($sql) = @_; + my $count = $node->safe_psql('postgres', + "SELECT analyze_count FROM pg_stat_all_tables_internal WHERE relname = 'vstat_barrier'"); + BAIL_OUT("missing local ANALYZE count for vstat_barrier") + unless $count =~ /^\d+$/; + my ($result, $stdout, $stderr) = $node->psql('postgres', + "$sql;\nANALYZE vstat_barrier;", on_error_die => 1); + wait_for_stats( + "SELECT analyze_count = $count + 1 FROM pg_stat_all_tables_internal WHERE relname = 'vstat_barrier'", + "collector report after $sql"); + return $stderr; +} + +sub wait_for_updates +{ + my ($table, $count) = @_; + wait_for_stats( + "SELECT n_tup_upd = $count FROM pg_stat_all_tables_internal WHERE relname = '$table'", + "UPDATE report for $table"); +} + +sub populated_pages +{ + my ($table, $predicate) = @_; + $predicate ||= 'true'; + return $node->safe_psql('postgres', + "SELECT count(DISTINCT split_part(ctid::text, ',', 1)) FROM $table WHERE $predicate"); +} + +sub vacuum_table +{ + my ($table, $options, $settings) = @_; + $options ||= ''; + $settings ||= ''; + return run_and_wait("$settings VACUUM $options $table"); +} + +sub counters +{ + my ($table, $columns) = @_; + $columns ||= $all_counters; + return $node->safe_psql('postgres', + "SELECT $columns FROM pg_stat_vacuum_tables WHERE relname = '$table'"); +} + +sub index_counters +{ + my ($index, $columns) = @_; + $columns ||= $all_counters; + return $node->safe_psql('postgres', + "SELECT $columns FROM pg_stat_vacuum_indexes WHERE indexrelname = '$index'"); +} + +sub database_counters +{ + my ($columns) = @_; + $columns ||= $all_counters; + return $node->safe_psql('postgres', + "SELECT $columns FROM pg_stat_vacuum_database WHERE datname = 'postgres'"); +} + +# Read the same statistics snapshot before measuring its hash allocations. +sub snapshot_bytes +{ + return $node->safe_psql('postgres', q{ +BEGIN; +DO $$ BEGIN PERFORM count(*) FROM pg_stat_all_tables_internal WHERE n_tup_ins >= 0; END $$; +SELECT sum(used_bytes) FROM pg_backend_memory_contexts +WHERE name IN ('Databases hash', 'Per-database table'); +COMMIT; +}); +} + +subtest 'tracking disabled and enabled' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_off (id int) WITH (autovacuum_enabled = off); +INSERT INTO vstat_off SELECT generate_series(1, 1000); +DELETE FROM vstat_off; +}); + my $verbose = vacuum_table('vstat_off', 'VERBOSE'); + like($verbose, qr/table "vstat_off": vacuum statistics.*?cost-based delay: 0\.000 ms/s, + 'VERBOSE reports measurements even with statistics tracking disabled'); + is(counters('vstat_off', $vacuum_zero), 't', + 'extended counters stay zero while tracking is disabled'); + is(counters('vstat_off', 'total_vacuum_time > 0, total_autovacuum_time = 0'), + 't|t', 'core timing remains enabled independently of extended counters'); + is($node->safe_psql('postgres', + "SELECT vacuum_count FROM pg_stat_all_tables_internal WHERE relname = 'vstat_off'"), + '1', 'ordinary vacuum statistics are still collected'); + for my $am ('ao_row', 'ao_column') + { + my $table = "vstat_off_$am"; + $node->safe_psql('postgres', qq{ +CREATE TABLE $table (id int) USING $am; +CREATE INDEX ${table}_idx ON $table (id); +INSERT INTO $table SELECT generate_series(1, 10000); +DELETE FROM $table WHERE id % 2 = 0; +}); + my $verbose = vacuum_table($table, 'VERBOSE', q{ +SET track_cost_delay_timing = on; +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 1; +}); + is(counters($table, "$vacuum_zero AND total_file_segs = 0"), 't', + "$am extended counters and segment snapshot stay zero with tracking off"); + is(index_counters("${table}_idx", $vacuum_zero), 't', + "$am index work counters stay zero with tracking off"); + is(counters($table, 'total_vacuum_time > 0 AND total_vacuum_delay_time > 0'), + 't', "$am table timing remains independent of extended tracking"); + is(index_counters("${table}_idx", 'total_vacuum_time > 0 AND total_vacuum_delay_time > 0'), + 't', "$am compaction index timing remains independent of extended tracking"); + like($verbose, qr/[1-9]\d* bytes truncated; \d+ file segments remain\./, + "$am VERBOSE retains byte and segment measurements with tracking off"); + + # The following run has no obsolete segments and uses scan_index(). + my $index_time = index_counters("${table}_idx", 'total_vacuum_time'); + vacuum_table($table); + is(index_counters("${table}_idx", "total_vacuum_time > $index_time"), 't', + "$am cleanup-only index timing advances with extended tracking off"); + is(index_counters("${table}_idx", $vacuum_zero), 't', + "$am cleanup-only index work counters remain disabled"); + } + is(database_counters($vacuum_zero), 't', + 'heap and AO work do not populate extended database totals with tracking off'); + # VM changes are ordinary DML statistics even with vacuum tracking off. + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_vm_off (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_vm_off SELECT generate_series(1, 1000); +}); + vacuum_table('vstat_vm_off', 'FREEZE'); + my $pages = $node->safe_psql('postgres', + "SELECT pg_relation_size('vstat_vm_off') / current_setting('block_size')::bigint"); + my ($db_frozen, $db_visible) = split /\|/, database_counters($vm_columns); + $node->safe_psql('postgres', 'UPDATE vstat_vm_off SET id = id + 1000'); + wait_for_updates('vstat_vm_off', 1000); + is(counters('vstat_vm_off', $vm_columns), "$pages|$pages", + 'DML counts exact VM clearings while vacuum tracking is off'); + is(database_counters("frozen_page_marks_cleared - $db_frozen, visible_page_marks_cleared - $db_visible"), + "$pages|$pages", 'database VM totals receive the same disabled-tracking DML report'); + is(counters('vstat_vm_off', $vacuum_zero), 't', + 'VM collection does not enable vacuum counters'); + $node->restart; + is(counters('vstat_vm_off', $vm_columns), "$pages|$pages", + 'VM counters survive a clean restart with tracking off'); + # dynahash allocates entries in batches. With only a few relations, + # a larger entry can use a slightly smaller batch and appear cheaper. + # Populate enough ordinary entries to exceed that allocation rounding; + # none of these relations has been vacuumed. + $node->safe_psql('postgres', q{ +CREATE SCHEMA vstat_memory; +DO $$ BEGIN + FOR i IN 1..1024 LOOP + EXECUTE format('CREATE TABLE vstat_memory.t%s (id int) WITH (autovacuum_enabled = off)', i); + EXECUTE format('INSERT INTO vstat_memory.t%s VALUES (1)', i); + END LOOP; +END $$; +}); + wait_for_stats(q{ +SELECT count(*) = 1024 AND bool_and(n_tup_ins = 1) +FROM pg_stat_all_tables_internal WHERE schemaname = 'vstat_memory' +}, 'ordinary statistics for the memory-allocation test'); + my $off_bytes = snapshot_bytes(); + is($node->safe_psql('postgres', + "SELECT context FROM pg_settings WHERE name = 'track_vacuum_statistics'"), + 'postmaster', 'tracking is fixed at server startup'); + my ($result, $stdout, $stderr) = $node->psql('postgres', + 'SET track_vacuum_statistics = on'); + is($result, 3, 'a session cannot enable tracking'); + like($stderr, qr/cannot be changed without restarting the server/, + 'the error explains that a restart is required'); + $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = on'); + $node->reload; + wait_for_stats(q{SELECT pending_restart FROM pg_settings WHERE name = 'track_vacuum_statistics'}, + 'configuration reload to notice the startup setting'); + is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'off', + 'reload leaves tracking disabled'); + $node->restart; + is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'on', + 'restart enables tracking'); + is(counters('vstat_vm_off', $vm_columns), "$pages|$pages", + 'enabling vacuum tracking preserves VM counters collected while off'); + is(counters('vstat_off', $vacuum_zero), 't', + 'new vacuum blocks start at zero when reading ordinary-only statistics'); + cmp_ok(snapshot_bytes(), '>', $off_bytes, + 'enabled snapshots allocate vacuum counters with ordinary relation and DB entries'); + +}; + +subtest 'removed tuples and truncated pages' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_heap (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_heap SELECT generate_series(1, 10000); +DELETE FROM vstat_heap WHERE id % 2 = 0; +}); + is(counters('vstat_heap', $all_zero), 't', + 'all counters are zero before the first vacuum'); + # Heap insertion can pre-extend the relation with empty tail pages. + # Disable truncation in this run to make its zero counter deterministic. + my $verbose = vacuum_table('vstat_heap', '(VERBOSE, TRUNCATE false)'); + is(counters('vstat_heap', 'tuples_deleted, dead_tuples, pages_deleted, dead_pages'), + '5000|0|0|0', 'vacuum removes exactly half the rows without truncation'); + like($verbose, qr/pages with dead tuples not yet removable: 0\n/, + 'VERBOSE reports zero pages retaining dead tuples'); + like($verbose, qr/pages with tuples frozen: 0\n/, + 'VERBOSE reports zero pages frozen with the default freeze age'); + my $visible = counters('vstat_heap', 'pages_all_visible'); + like($verbose, qr/pages marked all-visible: \Q$visible\E\n/, + 'VERBOSE all-visible count matches the first vacuum report'); + like($verbose, qr/scanned index "vstat_heap_pkey".*?cost-based delay: 0\.000 ms/s, + 'VERBOSE reports the index cost delay'); + is(index_counters('vstat_heap_pkey', 'tuples_deleted, pages_deleted'), + '5000|0', 'every other index key remains; no index pages are deleted'); + is(counters('vstat_heap', 'total_vacuum_time > 0, total_autovacuum_time = 0, total_vacuum_delay_time = 0'), + 't|t|t', 'heap vacuum takes time but has no cost delay'); + is(index_counters('vstat_heap_pkey', 'total_vacuum_time > 0, total_autovacuum_time = 0, total_vacuum_delay_time = 0'), + 't|t|t', 'index vacuum takes time but has no cost delay'); + + $node->safe_psql('postgres', 'DELETE FROM vstat_heap'); + my $pages_before = $node->safe_psql('postgres', + "SELECT pg_relation_size('vstat_heap') / current_setting('block_size')::bigint"); + my $quiet = vacuum_table('vstat_heap'); + unlike($quiet, qr/vacuum statistics|pages marked all-visible|cost-based delay/, + 'ordinary VACUUM does not emit VERBOSE statistics at the default message level'); + is(counters('vstat_heap', 'tuples_deleted, dead_tuples, pages_deleted'), + "10000|0|$pages_before", 'the second vacuum removes the remaining rows and truncates every heap page'); + is(counters('vstat_heap', "bytes_removed = pages_deleted * current_setting('block_size')::bigint"), + 't', 'heap truncation reports exact bytes'); + is(index_counters('vstat_heap_pkey', 'tuples_deleted, pages_deleted > 0'), + '10000|t', 'index tuple and page deletion counters accumulate'); +}; + +subtest 'cluster views return local rows once in utility mode' => sub { + is($node->safe_psql('postgres', 'SHOW gp_role'), 'utility', + 'this scenario uses a direct utility connection'); + for my $view ( + ['tables', 'relid, schemaname, relname', "relname = 'vstat_heap'"], + ['indexes', 'relid, indexrelid, schemaname, relname, indexrelname', + "relname = 'vstat_heap'"], + ['database', 'datid, datname', "datname = 'postgres'"]) + { + my ($suffix, $keys, $filter) = @$view; + my $columns = "$keys, $all_counters"; + $columns .= ', total_analyze_time, total_autoanalyze_time' if $suffix eq 'tables'; + $columns .= ', total_file_segs' if $suffix eq 'tables'; + $columns .= ', vacuum_interrupt_count' if $suffix eq 'database'; + my $local = "SELECT $columns FROM pg_stat_vacuum_$suffix WHERE $filter"; + my $cluster = "SELECT $columns FROM gp_stat_vacuum_$suffix WHERE $filter"; + is($node->safe_psql('postgres', qq{ +SELECT count(*) = 1 AND bool_and(gp_segment_id = gp_execution_segment()) +FROM gp_stat_vacuum_$suffix WHERE $filter +}), 't', "$suffix view returns one row identified by the connected node"); + is($node->safe_psql('postgres', qq{ +SELECT NOT EXISTS ( + ($cluster EXCEPT ALL $local) + UNION ALL + ($local EXCEPT ALL $cluster) +) +}), 't', "$suffix cluster and local views have identical rows and counters"); + } +}; + +subtest 'a repeatable-read snapshot prevents removal' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_snapshot (id int PRIMARY KEY, val int) + WITH (autovacuum_enabled = off); +INSERT INTO vstat_snapshot SELECT i, i FROM generate_series(1, 1000) g(i); +}); + my $dead_pages = populated_pages('vstat_snapshot', 'id > 900'); + my ($in, $out) = ('', ''); + my $timer = IPC::Run::timeout($TestLib::timeout_default); + my $reader = $node->background_psql('postgres', \$in, \$out, $timer); + $out = ''; + $in = "BEGIN ISOLATION LEVEL REPEATABLE READ;\n" + . "SELECT count(*) FROM vstat_snapshot;\n\\echo snapshot_ready\n"; + pump_until($reader, $timer, \$out, qr/^snapshot_ready\r?$/m) + or BAIL_OUT('reader did not acquire its snapshot'); + like($out, qr/^1000\r?$/m, 'reader sees all original rows'); + + # Updating the indexed column prevents HOT, making index removals exact. + $node->safe_psql('postgres', 'UPDATE vstat_snapshot SET id = id + 1000 WHERE id > 900'); + my $verbose = vacuum_table('vstat_snapshot', 'VERBOSE'); + is(counters('vstat_snapshot', 'tuples_deleted, dead_tuples, dead_pages, pages_frozen'), + "0|100|$dead_pages|0", '100 old tuple versions remain on exactly the affected pages'); + like($verbose, qr/pages with dead tuples not yet removable: \Q$dead_pages\E\n/, + 'VERBOSE reports the pages held back by the snapshot'); + is(index_counters('vstat_snapshot_pkey', 'tuples_deleted'), + '0', 'index entries needed by the reader are retained'); + + $in = "COMMIT;\n\\q\n"; + $reader->finish; + vacuum_table('vstat_snapshot'); + is(counters('vstat_snapshot', 'tuples_deleted, dead_tuples, pages_frozen'), + '100|100|0', 'after commit, 100 versions are removed; cumulative dead_tuples stays at 100'); + is(index_counters('vstat_snapshot_pkey', 'tuples_deleted'), '100', + 'after commit, the index removes exactly the 100 obsolete entries'); +}; + +subtest 'heap page reports and core VM counters in extension views' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_vm (id int PRIMARY KEY, val int) + WITH (autovacuum_enabled = off, fillfactor = 50); +INSERT INTO vstat_vm SELECT i, i FROM generate_series(1, 5000) g(i); +}); + my $pages = populated_pages('vstat_vm'); + my $verbose = vacuum_table('vstat_vm', 'VERBOSE', + 'SET vacuum_freeze_min_age = 1000000000; SET vacuum_freeze_table_age = 1000000000;'); + is(counters('vstat_vm', 'pages_frozen, pages_all_visible, frozen_page_marks_cleared, visible_page_marks_cleared'), + "0|$pages|0|0", 'ordinary vacuum marks each populated page visible without freezing it'); + like($verbose, qr/pages with tuples frozen: 0\n/, + 'VERBOSE reports no freezing with a high freeze age'); + like($verbose, qr/pages marked all-visible: \Q$pages\E\n/, + 'VERBOSE reports the exact number of newly visible pages'); + + # Leave enough room for new versions on the same pages and prevent HOT. + # The distinct counter values detect swapped columns in extension views. + $node->safe_psql('postgres', 'UPDATE vstat_vm SET id = id + 10000'); + wait_for_updates('vstat_vm', 5000); + # Do not scan the heap here: that could prune the obsolete versions before + # VACUUM gets to count their removal. + is($node->safe_psql('postgres', + "SELECT pg_relation_size('vstat_vm') / current_setting('block_size')::bigint"), + $pages, 'updated versions fit on the original pages'); + is(counters('vstat_vm', 'frozen_page_marks_cleared, visible_page_marks_cleared'), + "0|$pages", 'extension view reports only all-visible clearings for unfrozen pages'); + + $verbose = vacuum_table('vstat_vm', '(FREEZE, VERBOSE)'); + my $visible = 2 * $pages; + is(counters('vstat_vm', 'tuples_deleted, pages_frozen, pages_all_visible'), + "5000|$pages|$visible", 'FREEZE reports removals, freezing and restored visibility separately'); + is(counters('vstat_vm', 'frozen_page_marks_cleared, visible_page_marks_cleared'), + "0|$pages", 'restoring VM flags adds no clearings'); + like($verbose, qr/pages with tuples frozen: \Q$pages\E\n/, + 'VERBOSE reports exactly one freeze per populated page'); + like($verbose, qr/pages marked all-visible: \Q$pages\E\n/, + 'VERBOSE reports this run\'s visibility work, not the cumulative total'); + is(counters('vstat_vm', 'freeze_age_vacuum_count'), '1', + 'FREEZE counts the vacuum made aggressive by the freeze age'); + like($verbose, qr/aggressive scan required by freeze age: yes/, + 'VERBOSE explains freeze-age-driven aggressive scanning'); + + $verbose = vacuum_table('vstat_vm', '(FREEZE, VERBOSE)'); + is(counters('vstat_vm', 'pages_frozen, pages_all_visible'), + "$pages|$visible", 'another FREEZE adds no freezing or visibility work'); + like($verbose, qr/pages with tuples frozen: 0\n/, + 'VERBOSE reports no repeated freezing'); + like($verbose, qr/pages marked all-visible: 0\n/, + 'VERBOSE reports no repeated visibility changes'); + + $node->safe_psql('postgres', 'DELETE FROM vstat_vm'); + wait_for_stats(q{ +SELECT n_tup_del = 5000 FROM pg_stat_all_tables_internal WHERE relname = 'vstat_vm' +}, 'DELETE report for frozen pages'); + is(counters('vstat_vm', 'frozen_page_marks_cleared, visible_page_marks_cleared'), + "$pages|$visible", 'extension view exposes both core VM counters with distinct totals'); +}; + +subtest 'append-optimized compaction' => sub { + for my $am ('ao_row', 'ao_column') + { + my $table = "vstat_$am"; + $node->safe_psql('postgres', qq{ +CREATE TABLE $table (id int, val int) USING $am; +CREATE INDEX ${table}_idx ON $table (id); +INSERT INTO $table SELECT i, i FROM generate_series(1, 10000) g(i); +DELETE FROM $table WHERE id % 2 = 0; +}); + # Preserve the zero-freeze-age case: AO still must not count a + # freeze-age vacuum, unlike its auxiliary heap relations. + my $verbose = vacuum_table($table, 'VERBOSE', 'SET vacuum_freeze_table_age = 0;'); + is(counters($table, 'tuples_deleted, dead_tuples, pages_frozen, pages_all_visible, freeze_age_vacuum_count'), + '5000|0|0|0|0', "$am compaction removes 5000 rows and has no heap VM or freezing work"); + # Compaction relocates surviving tuples, so all original index + # entries, including those of surviving rows, become obsolete. + is(index_counters("${table}_idx", 'tuples_deleted'), + '10000', "$am index cleanup removes the old TIDs"); + is(counters($table, 'total_vacuum_time > 0, total_autovacuum_time = 0, total_vacuum_delay_time = 0'), + 't|t|t', "$am compaction takes time without cost delay"); + my $bytes = counters($table, 'bytes_removed'); + my $segrel = $node->safe_psql('postgres', + "SELECT segrelid::regclass FROM pg_appendonly WHERE relid = '$table'::regclass"); + my $segs = $node->safe_psql('postgres', "SELECT count(*) FROM $segrel"); + is(counters($table, 'total_file_segs'), $segs, + "$am reports the remaining segment metadata entries"); + is($node->safe_psql('postgres', + "SELECT total_file_segs FROM gp_stat_vacuum_tables WHERE relname = '$table'"), + $segs, "$am cluster view exposes the segment snapshot in utility mode"); + like($verbose, qr/\Q$bytes bytes truncated; $segs file segments remain.\E/, + "$am VERBOSE byte and segment counts match the report"); + like($verbose, + qr/append-optimized table "\Q$table\E": vacuum statistics\nDETAIL: \Q0 dead tuples remain.\E\n\d+ bytes truncated; \d+ file segments remain\.\nelapsed: \d+\.\d{3} ms, cost-based delay: 0\.000 ms/, + "$am VERBOSE reports remaining dead tuples and accumulated phase time"); + is(counters($table, 'pages_deleted > 0, dead_pages, frozen_page_marks_cleared, visible_page_marks_cleared'), + 't|0|0|0', "$am reports freed space without heap page or VM counters"); + + # Without obsolete segment files, AO still runs index cleanup. + # Its time must be reported even if the AM returns no page statistics. + my $index_time = index_counters("${table}_idx", 'total_vacuum_time'); + vacuum_table($table); + is(counters($table, 'bytes_removed, total_file_segs'), "$bytes|$segs", + "$am idle vacuum neither adds reclaimed bytes nor sums segment counts"); + is(index_counters("${table}_idx", "tuples_deleted, total_vacuum_time > $index_time, total_vacuum_delay_time = 0"), + '10000|t|t', "$am reports index cleanup without deleting more TIDs"); + + $node->safe_psql('postgres', "DELETE FROM $table WHERE id <= 200"); + vacuum_table($table, '', 'SET gp_appendonly_compaction = off;'); + is(counters($table, 'tuples_deleted, dead_tuples'), + '5000|100', "$am counts hidden rows left when compaction is disabled"); + } +}; + +subtest 'AO truncate reports exact bytes beyond the 32-bit boundary' => sub { + my $block_size = $node->safe_psql('postgres', 'SHOW block_size'); + for my $am ('ao_row', 'ao_column') + { + my $table = "vstat_tail_$am"; + $node->safe_psql('postgres', qq{ +CREATE TABLE $table (id int) USING $am; +INSERT INTO $table SELECT generate_series(1, 10000); +CHECKPOINT; +}); + my $relpath = $node->safe_psql('postgres', + "SELECT pg_relation_filepath('$table')"); + my @files = grep { -f $_ && -s $_ } + glob($node->data_dir . '/' . $relpath . '*'); + my ($file) = grep { /\Q$relpath\E(?:\.\d+)?$/ } @files; + defined($file) or BAIL_OUT("no data file for $table"); + my $original_size = -s $file; + my $tail_size = 2**31 + 17; + # Model an aborted insert's tail in this disposable cluster with a + # sparse file. This exercises real truncate without writing 2 GiB. + open(my $fh, '+<', $file) or die "$file: $!"; + truncate($fh, $original_size + $tail_size) or die "truncate: $!"; + close($fh) or die "close: $!"; + vacuum_table($table); + is(-s $file, $original_size, "$am removes the physical tail"); + my $pages = int(($tail_size + $block_size - 1) / $block_size); + is(counters($table, 'bytes_removed, pages_deleted'), "$tail_size|$pages", + "$am preserves exact bytes and rounds the block equivalent up"); + my $segs = counters($table, 'total_file_segs'); + cmp_ok($segs, '>', 0, "$am has a segment snapshot"); + $node->restart; + is(counters($table, 'bytes_removed, pages_deleted, total_file_segs'), + "$tail_size|$pages|$segs", "$am counters and state survive restart"); + vacuum_table($table); + is(counters($table, 'bytes_removed, pages_deleted, total_file_segs'), + "$tail_size|$pages|$segs", "$am repeated vacuum does not recount the tail or segments"); + } +}; + +subtest 'SP-GiST does not recount reusable pages' => sub { + for my $am ('heap', 'ao_row', 'ao_column') + { + my $table = "vstat_spg_$am"; + $node->safe_psql('postgres', qq{ +CREATE TABLE $table (id int, p point) USING $am; +CREATE INDEX ${table}_idx ON $table USING spgist (p); +INSERT INTO $table SELECT i, point(i,i) FROM generate_series(1, 10000) g(i); +DELETE FROM $table; +}); + vacuum_table($table, '(INDEX_CLEANUP ON)'); + my $pages = index_counters("${table}_idx", 'pages_deleted'); + is(index_counters("${table}_idx", 'tuples_deleted'), '10000', + "$am SP-GiST removes the original index entries"); + vacuum_table($table, '(INDEX_CLEANUP ON)'); + is(index_counters("${table}_idx", 'pages_deleted'), $pages, + "$am SP-GiST does not count already empty pages again"); + is(index_counters("${table}_idx", 'bytes_removed'), '0', + "$am SP-GiST page reuse returns no bytes to the filesystem"); + } +}; + +subtest 'VACUUM FULL leaves the extended counters unchanged' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_full (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_full SELECT generate_series(1, 10000); +DELETE FROM vstat_full WHERE id % 2 = 0; +}); + vacuum_table('vstat_full'); + is(counters('vstat_full', 'tuples_deleted'), '5000', + 'plain vacuum establishes nonzero counters'); + $node->safe_psql('postgres', 'DELETE FROM vstat_full'); + wait_for_stats( + "SELECT n_tup_del = 10000 FROM pg_stat_all_tables_internal WHERE relname = 'vstat_full'", + 'DELETE report before VACUUM FULL'); + my $table_before = counters('vstat_full'); + my $index_before = index_counters('vstat_full_pkey'); + run_and_wait('VACUUM FULL vstat_full'); + is(counters('vstat_full'), $table_before, 'VACUUM FULL leaves table counters unchanged'); + is(index_counters('vstat_full_pkey'), $index_before, 'VACUUM FULL leaves index counters unchanged'); +}; + +subtest 'cost-based delay is part of total time' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_delay (id int) WITH (autovacuum_enabled = off); +INSERT INTO vstat_delay SELECT generate_series(1, 10000); +DELETE FROM vstat_delay; +}); + # Session-local settings do not leak into later scenarios. + my $verbose = vacuum_table('vstat_delay', 'VERBOSE', + 'SET track_cost_delay_timing = on; SET vacuum_cost_delay = 1; SET vacuum_cost_limit = 1;'); + is(counters('vstat_delay', 'total_vacuum_delay_time > 0 AND total_vacuum_delay_time <= total_vacuum_time'), + 't', 'the measured cost delay is positive and included in total time'); + my $delay = sprintf('%.3f', counters('vstat_delay', 'total_vacuum_delay_time')); + like($verbose, qr/elapsed: \d+\.\d{3} ms, cost-based delay: \Q$delay\E ms/, + 'VERBOSE reports elapsed time and the same measured cost delay'); +}; + +subtest 'index pages are not counted again by later vacuums' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_idx (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_idx SELECT generate_series(1, 100000); +DELETE FROM vstat_idx WHERE id <= 90000; +}); + vacuum_table('vstat_idx'); + my $pages = index_counters('vstat_idx_pkey', 'pages_deleted'); + cmp_ok($pages, '>', 0, 'the first vacuum deleted index pages'); + $node->safe_psql('postgres', 'DELETE FROM vstat_idx WHERE id > 99000'); + vacuum_table('vstat_idx'); + is(index_counters('vstat_idx_pkey', 'tuples_deleted'), '91000', + 'the next vacuum removed exactly 1000 more index entries'); + is(index_counters('vstat_idx_pkey', "pages_deleted >= $pages AND pages_deleted - $pages < $pages / 2"), + 't', 'deleted-page totals accumulate without recounting the first run'); +}; + +subtest 'statistics snapshots release their vacuum counters' => sub { + # Keep the last snapshot alive until the context count is read. + # Outside this transaction its context would already have been freed. + is($node->safe_psql('postgres', q{ +BEGIN; +DO $$ +BEGIN + FOR i IN 1..50 LOOP + PERFORM pg_stat_clear_snapshot(); + PERFORM count(*) FROM pg_stat_vacuum_tables WHERE tuples_deleted > 0; + END LOOP; +END +$$; +SELECT count(*) FROM pg_backend_memory_contexts +WHERE name = 'Databases hash'; +COMMIT; +}), '1', 'only the current snapshot owns statistics entries with vacuum counters'); +}; + +subtest 'disabling tracking omits vacuum counters and preserves ordinary statistics' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_startup (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_startup SELECT generate_series(1, 1000); +}); + vacuum_table('vstat_startup', 'FREEZE'); + $node->safe_psql('postgres', 'UPDATE vstat_startup SET id = id + 1000'); + wait_for_updates('vstat_startup', 1000); + vacuum_table('vstat_startup', 'FREEZE'); + is(counters('vstat_startup', + 'tuples_deleted, frozen_page_marks_cleared > 0, visible_page_marks_cleared > 0'), + '1000|t|t', 'populate vacuum and VM revision counters before disabling'); + my $ordinary_query = "SELECT vacuum_count, n_tup_upd FROM pg_stat_all_tables_internal WHERE relname = 'vstat_startup'"; + my $ordinary_before = $node->safe_psql('postgres', $ordinary_query); + my $vm_before = counters('vstat_startup', $vm_columns); + my $timing_columns = 'total_vacuum_time, total_autovacuum_time, total_vacuum_delay_time, total_autovacuum_delay_time'; + my $timing_before = counters('vstat_startup', $timing_columns); + my $db_vm_before = database_counters($vm_columns); + my $on_bytes = snapshot_bytes(); + $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = off'); + $node->restart; + is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'off', + 'restart disables tracking'); + is(counters('vstat_startup', $vm_columns), $vm_before, + 'disabling vacuum tracking preserves relation VM counters'); + is(counters('vstat_startup', $timing_columns), $timing_before, + 'disabling extended tracking preserves core vacuum timing'); + is(database_counters($vm_columns), $db_vm_before, + 'disabling vacuum tracking preserves database VM totals'); + is(counters('vstat_startup', $vacuum_zero), 't', 'disabled table counters read as zero'); + is(index_counters('vstat_startup_pkey', $vacuum_zero), 't', 'disabled index counters read as zero'); + is(database_counters($vacuum_zero), 't', 'disabled database counters read as zero'); + is($node->safe_psql('postgres', 'SELECT bool_and(total_file_segs = 0) FROM pg_stat_vacuum_tables'), + 't', 'disabled AO segment snapshots read as zero'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'reading statistics with tracking disabled preserves ordinary counters'); + cmp_ok(snapshot_bytes(), '<', $on_bytes, + 'disabled snapshots omit vacuum storage but retain ordinary VM counters'); + $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = on'); + $node->restart; + is(counters('vstat_startup', $vacuum_zero), 't', + 're-enabling does not restore discarded table vacuum counters'); + is(database_counters($vacuum_zero), 't', + 're-enabling does not restore discarded database vacuum counters'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'ordinary statistics survive both changes in record size'); + vacuum_table('vstat_startup', 'FREEZE'); + is(counters('vstat_startup', 'total_vacuum_time > 0'), 't', + 'vacuum reporting resumes after tracking is enabled again'); +}; + +subtest 'clean restart preserves statistics; crash recovery resets them' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_restart (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_restart SELECT generate_series(1, 1000); +DELETE FROM vstat_restart WHERE id % 2 = 0; +}); + vacuum_table('vstat_restart', 'FREEZE'); + is(counters('vstat_restart', 'tuples_deleted'), '500', 'table counters are populated before restart'); + is(index_counters('vstat_restart_pkey', 'tuples_deleted'), '500', 'index counters are populated before restart'); + my $table_before = counters('vstat_restart'); + my $index_before = index_counters('vstat_restart_pkey'); + my $db_before = database_counters(); + $node->restart; + is(counters('vstat_restart'), $table_before, 'table counters survive a clean restart'); + is(index_counters('vstat_restart_pkey'), $index_before, 'index counters survive a clean restart'); + is(database_counters(), $db_before, 'database counters survive a clean restart'); + $node->stop('immediate'); + $node->start; + is(counters('vstat_restart', $all_zero), 't', 'crash recovery resets table counters'); + is(index_counters('vstat_restart_pkey', $all_zero), 't', 'crash recovery resets index counters'); + is(database_counters($all_zero), 't', 'crash recovery resets database counters'); +}; + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/t/002_index_vacuum_time.pl b/contrib/vacuum_stats/t/002_index_vacuum_time.pl new file mode 100644 index 00000000000..ff41161e7fa --- /dev/null +++ b/contrib/vacuum_stats/t/002_index_vacuum_time.pl @@ -0,0 +1,252 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Test the total_vacuum_time counter of pg_stat_vacuum_indexes: the time vacuum +# spent processing each index, accumulated over bulkdelete and cleanup passes, +# mirroring the table-level counter in pg_stat_vacuum_tables. + +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('main'); +$node->init; +$node->append_conf( + 'postgresql.conf', qq[ +autovacuum = off +track_cost_delay_timing = on +]); +$node->start; +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats'); + +# The index has to span enough pages for its vacuum passes to exceed +# vacuum_cost_limit on buffer hits alone (its pages are still dirty from the +# load, so dirtying them costs nothing). Scale the row count with the block +# size, so that builds with larger blocks (32kB in Cloudberry) keep the page +# count of the default 8kB. +my $nrows = 100000 * + ($node->safe_psql('postgres', 'SHOW block_size') / 8192); + +# Enough rows that the bulkdelete pass over the index takes measurable time; +# a small vacuum_cost_delay adds deterministic delay time on fast machines. +$node->safe_psql( + 'postgres', qq[ +CREATE TABLE vactime_t (id int PRIMARY KEY, v text) WITH (autovacuum_enabled = off); +INSERT INTO vactime_t SELECT g, repeat('x', 10) FROM generate_series(1, $nrows) g; +DELETE FROM vactime_t WHERE id % 2 = 0; +]); +$node->safe_psql( + 'postgres', qq[ +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 200; +VACUUM vactime_t; +]); + +# The collector receives the ordinary vacuum and index-pass reports asynchronously. +$node->poll_query_until('postgres', q{ +SELECT total_vacuum_time > 0 AND total_vacuum_delay_time > 0 +FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey' +}) or BAIL_OUT('index timing report did not reach the collector'); +$node->poll_query_until('postgres', q{ +SELECT total_vacuum_time > 0 AND total_vacuum_delay_time > 0 +FROM pg_stat_vacuum_database WHERE datname = current_database() +}) or BAIL_OUT('database timing report did not reach the collector'); + +# The upper bound guards against garbage such as an epoch-based elapsed time +# leaking into the counter. +is( $node->safe_psql( + 'postgres', qq[ +SELECT total_vacuum_time > 0 AND total_vacuum_time < 600000 + FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey']), + 't', + 'total_vacuum_time advanced sanely for the index in pg_stat_vacuum_indexes'); + +is( $node->safe_psql( + 'postgres', qq[ +SELECT total_vacuum_delay_time > 0 AND total_vacuum_delay_time <= total_vacuum_time + FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey']), + 't', + 'total_vacuum_delay_time advanced and not above total_vacuum_time'); + +is( $node->safe_psql( + 'postgres', qq[ +SELECT total_autovacuum_time = 0 FROM pg_stat_vacuum_indexes + WHERE indexrelname = 'vactime_t_pkey']), + 't', + 'manual vacuum did not count into total_autovacuum_time'); + +# The same run must have accumulated into the database-wide totals. +is( $node->safe_psql( + 'postgres', qq[ +SELECT total_vacuum_time > 0 AND total_vacuum_time < 600000 + AND total_vacuum_delay_time > 0 + AND total_vacuum_delay_time <= total_vacuum_time + FROM pg_stat_vacuum_database WHERE datname = current_database()]), + 't', + 'database-wide vacuum times advanced sanely in pg_stat_vacuum_database'); + +# The database total is the sum of table runs, which already include index +# work. Adding the positive per-index time again would break this equality. +# Cast each millisecond value to numeric before summing to avoid floating-point +# rounding differences. Shared relations belong to the database with OID zero. +is($node->safe_psql('postgres', q{ +SELECT d.total_vacuum_time::numeric = t.elapsed + AND d.total_vacuum_delay_time::numeric = t.delay +FROM pg_stat_vacuum_database d CROSS JOIN ( + SELECT sum(s.total_vacuum_time::numeric) AS elapsed, + sum(s.total_vacuum_delay_time::numeric) AS delay + FROM pg_stat_vacuum_tables s JOIN pg_class c ON c.oid = s.relid + WHERE NOT c.relisshared +) t +WHERE d.datname = current_database() +}), 't', 'database totals include index work exactly once'); + +# Turning off delay timing must preserve the accumulated delay while elapsed +# time keeps advancing. Keep cost delays enabled to exercise the timing GUC. +my $manual_times = q{ +SELECT total_vacuum_time, total_vacuum_delay_time +FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey' +}; +my ($elapsed_before, $delay_before) = + split /\|/, $node->safe_psql('postgres', $manual_times); +$node->safe_psql('postgres', q{ +SET track_cost_delay_timing = off; +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 200; +DELETE FROM vactime_t WHERE id % 4 = 1; +VACUUM (INDEX_CLEANUP ON) vactime_t; +}); +$node->poll_query_until('postgres', + "SELECT total_vacuum_time > $elapsed_before FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey'") + or BAIL_OUT('index elapsed time did not advance with delay timing disabled'); +my ($elapsed_after, $delay_after) = + split /\|/, $node->safe_psql('postgres', $manual_times); +is($delay_after, $delay_before, 'disabled delay timing preserves the index delay total'); + +# Exercise ANALYZE and VACUUM in the same backend, where the process-local +# delay accumulator survives between statements. Check their SQL totals +# independently rather than inferring one from the other. +# Drain earlier asynchronous vacuum reports before checking what ANALYZE +# alone changes. +$node->restart; +my $table_manual_times = q{ +SELECT total_vacuum_time, total_vacuum_delay_time +FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t' +}; +my $db_manual_times = q{ +SELECT total_vacuum_time, total_vacuum_delay_time +FROM pg_stat_vacuum_database WHERE datname = current_database() +}; +my @timing_queries = ($table_manual_times, $manual_times, $db_manual_times); +my @timing_labels = ('table', 'index', 'database'); +my @before_analyze = map { $node->safe_psql('postgres', $_) } @timing_queries; +$node->safe_psql('postgres', q{ +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 200; +ANALYZE vactime_t; +}); +$node->poll_query_until('postgres', q{ +SELECT total_analyze_time > 0 FROM pg_stat_vacuum_tables +WHERE relname = 'vactime_t' +}) or BAIL_OUT('ANALYZE timing report did not reach the collector'); +for my $i (0..2) +{ + is($node->safe_psql('postgres', $timing_queries[$i]), $before_analyze[$i], + "ANALYZE preserves $timing_labels[$i] vacuum elapsed and delay times"); +} +my $analyze_before = $node->safe_psql('postgres', q{ +SELECT total_analyze_time FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t' +}); +my ($table_elapsed, $table_delay) = split /\|/, $before_analyze[0]; +$node->safe_psql('postgres', q{ +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 200; +ANALYZE vactime_t; +VACUUM (ANALYZE, DISABLE_PAGE_SKIPPING, INDEX_CLEANUP ON) vactime_t; +ANALYZE vactime_t; +}); +$node->poll_query_until('postgres', qq{ +SELECT total_vacuum_time > $table_elapsed + AND total_vacuum_delay_time > $table_delay + AND total_analyze_time > $analyze_before +FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t' +}) or BAIL_OUT('VACUUM ANALYZE did not report both maintenance phases'); +is($node->safe_psql('postgres', q{ +SELECT total_vacuum_delay_time <= total_vacuum_time + AND total_autovacuum_time = 0 AND total_autoanalyze_time = 0 +FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t' +}), 't', 'VACUUM ANALYZE records separate manual timing totals'); + +# Cloudberry autovacuums catalogs. Churn comments to create dead catalog +# tuples and index entries, then wait for a real worker to process them. +$node->safe_psql('postgres', q{ +CREATE TABLE vactime_autovacuum (id int) WITH (autovacuum_enabled = off); +COMMENT ON TABLE vactime_autovacuum IS 'initial'; +VACUUM (INDEX_CLEANUP ON) pg_description; +}); +$node->poll_query_until('postgres', q{ +SELECT total_vacuum_time > 0 FROM pg_stat_vacuum_tables WHERE relname = 'pg_description' +}) or BAIL_OUT('catalog vacuum timing did not reach the collector'); +my $table_time = "SELECT total_vacuum_time, total_autovacuum_time FROM pg_stat_vacuum_tables WHERE relname = 'pg_description'"; +my $index_time = "SELECT total_vacuum_time, total_autovacuum_time FROM pg_stat_vacuum_indexes WHERE indexrelname = 'pg_description_o_c_o_index'"; +my $db_time = "SELECT total_vacuum_time, total_autovacuum_time FROM pg_stat_vacuum_database WHERE datname = current_database()"; +my @manual_before = map { (split /\|/, $node->safe_psql('postgres', $_))[0] } + ($table_time, $index_time, $db_time); +$node->safe_psql('postgres', q{ +DO $$ BEGIN + FOR i IN 1..200 LOOP + EXECUTE 'COMMENT ON TABLE vactime_autovacuum IS NULL'; + EXECUTE format('COMMENT ON TABLE vactime_autovacuum IS %L', i::text); + END LOOP; +END $$; +ALTER SYSTEM SET autovacuum_naptime = '1s'; +ALTER SYSTEM SET autovacuum_vacuum_threshold = 0; +ALTER SYSTEM SET autovacuum_vacuum_scale_factor = 0; +ALTER SYSTEM SET autovacuum_vacuum_insert_threshold = -1; +ALTER SYSTEM SET autovacuum = on; +}); +$node->reload; +$node->poll_query_until('postgres', q{ +SELECT total_autovacuum_time > 0 FROM pg_stat_vacuum_tables WHERE relname = 'pg_description' +}) or BAIL_OUT('autovacuum did not report catalog time'); +$node->poll_query_until('postgres', q{ +SELECT total_autovacuum_time > 0 FROM pg_stat_vacuum_indexes +WHERE indexrelname = 'pg_description_o_c_o_index' +}) or BAIL_OUT('autovacuum did not report catalog index time'); +$node->safe_psql('postgres', 'ALTER SYSTEM SET autovacuum = off'); +# Drain the collector on a clean shutdown before comparing exact totals. +# A worker disappearing from pg_stat_activity alone does not prove that +# its final UDP reports have reached the collector. +$node->restart; +my @auto_before; +my @queries = ($table_time, $index_time, $db_time); +my @labels = ('table', 'index', 'database'); +for my $i (0..2) +{ + my ($manual, $automatic) = split /\|/, $node->safe_psql('postgres', $queries[$i]); + is($manual, $manual_before[$i], "autovacuum preserves manual $labels[$i] time"); + cmp_ok($automatic, '>', 0, "autovacuum records separate $labels[$i] time"); + push @auto_before, $automatic; +} +$node->safe_psql('postgres', 'VACUUM (INDEX_CLEANUP ON) pg_description'); +$node->poll_query_until('postgres', + "SELECT total_vacuum_time > $manual_before[0] FROM pg_stat_vacuum_tables WHERE relname = 'pg_description'") + or BAIL_OUT('manual catalog vacuum did not report new time'); +for my $i (0..2) +{ + my ($manual, $automatic) = split /\|/, $node->safe_psql('postgres', $queries[$i]); + cmp_ok($manual, '>', $manual_before[$i], "manual vacuum adds $labels[$i] time"); + is($automatic, $auto_before[$i], "manual vacuum preserves autovacuum $labels[$i] time"); +} +my @before_restart = map { $node->safe_psql('postgres', $_) } @queries; +$node->restart; +for my $i (0..2) +{ + is($node->safe_psql('postgres', $queries[$i]), $before_restart[$i], + "both $labels[$i] timing counters survive restart"); +} + +$node->stop; + +done_testing(); diff --git a/contrib/vacuum_stats/t/003_vacuum_failsafe.pl b/contrib/vacuum_stats/t/003_vacuum_failsafe.pl new file mode 100644 index 00000000000..a41cd054b53 --- /dev/null +++ b/contrib/vacuum_stats/t/003_vacuum_failsafe.pl @@ -0,0 +1,113 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Test that vacuums entering the wraparound failsafe mode are counted in +# pg_stat_vacuum_tables.vacuum_failsafe_count and aggregated per database in +# pg_stat_vacuum_database.vacuum_failsafe_count. +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('main'); +$node->init; +# The failsafe cutoff is clamped to 1.05 * autovacuum_freeze_max_age, so use +# the minimum allowed value to keep the number of XIDs to burn small. +$node->append_conf( + 'postgresql.conf', qq[ +autovacuum = off +autovacuum_freeze_max_age = 100000 +]); +$node->start; +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats'); + +$node->safe_psql( + 'postgres', qq[ + CREATE TABLE tab_failsafe (i int); + INSERT INTO tab_failsafe SELECT generate_series(1, 100); +]); + +# A vacuum without failsafe pressure must not bump the counter. +$node->safe_psql('postgres', 'VACUUM FREEZE tab_failsafe;'); +$node->poll_query_until('postgres', q{ +SELECT vacuum_count = 1 FROM pg_stat_all_tables_internal WHERE relname = 'tab_failsafe' +}) or BAIL_OUT('normal vacuum report did not reach the collector'); +my $count = $node->safe_psql('postgres', + q[SELECT vacuum_failsafe_count FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe';] +); +is($count, '0', 'aggressive VACUUM FREEZE does not count as failsafe'); + +# Age the table past 1.05 * autovacuum_freeze_max_age: burn XIDs with +# aborted subtransactions (each aborted subxact consumes an assigned XID). +$node->safe_psql( + 'postgres', qq[ + CREATE TABLE burn_xids (i int); + DO \$\$ + BEGIN + FOR i IN 1..110000 LOOP + BEGIN + INSERT INTO burn_xids VALUES (1); + RAISE EXCEPTION 'burn'; + EXCEPTION WHEN OTHERS THEN + END; + END LOOP; + END \$\$; +]); + +# vacuum_failsafe_age = 0 makes the (clamped) cutoff kick in immediately. +$node->safe_psql( + 'postgres', qq[ + SET vacuum_failsafe_age = 0; + SET vacuum_multixact_failsafe_age = 0; + VACUUM tab_failsafe; +]); + +$node->poll_query_until('postgres', q{ +SELECT vacuum_failsafe_count = 1 FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe' +}) or BAIL_OUT('failsafe vacuum report did not reach the collector'); + +$count = $node->safe_psql('postgres', + q[SELECT vacuum_failsafe_count FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe';] +); +is($count, '1', 'failsafe vacuum counted in pg_stat_vacuum_tables'); + +my $db_count = $node->safe_psql('postgres', + q[SELECT vacuum_failsafe_count FROM pg_stat_vacuum_database WHERE datname = 'postgres';] +); +is($db_count, '1', 'failsafe vacuum counted in pg_stat_vacuum_database'); + +# Once the table has been frozen, another vacuum must not count the same +# failsafe event again. Wait for that run's report, not for an unchanged value. +$node->safe_psql('postgres', 'VACUUM tab_failsafe'); +$node->poll_query_until('postgres', q{ +SELECT vacuum_count = 3 FROM pg_stat_all_tables_internal WHERE relname = 'tab_failsafe' +}) or BAIL_OUT('subsequent vacuum report did not reach the collector'); +is($node->safe_psql('postgres', q{ +SELECT t.vacuum_failsafe_count, d.vacuum_failsafe_count +FROM pg_stat_vacuum_tables t CROSS JOIN pg_stat_vacuum_database d +WHERE t.relname = 'tab_failsafe' AND d.datname = current_database() +}), '1|1', 'subsequent ordinary vacuum does not count the failsafe again'); + +$node->restart; +is($node->safe_psql('postgres', q{ +SELECT t.vacuum_failsafe_count, d.vacuum_failsafe_count +FROM pg_stat_vacuum_tables t CROSS JOIN pg_stat_vacuum_database d +WHERE t.relname = 'tab_failsafe' AND d.datname = current_database() +}), '1|1', 'relation and database failsafe counts survive a clean restart'); + +$node->safe_psql('postgres', + q{SELECT pg_stat_reset_single_table_counters('tab_failsafe'::regclass)}); +$node->poll_query_until('postgres', q{ +SELECT vacuum_failsafe_count = 0 FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe' +}) or BAIL_OUT('relation failsafe counter was not reset'); +is($node->safe_psql('postgres', q{ +SELECT vacuum_failsafe_count FROM pg_stat_vacuum_database WHERE datname = current_database() +}), '1', 'resetting a relation preserves the database failsafe total'); + +$node->safe_psql('postgres', 'SELECT pg_stat_reset()'); +ok($node->poll_query_until('postgres', q{ +SELECT vacuum_failsafe_count = 0 FROM pg_stat_vacuum_database WHERE datname = current_database() +}), 'database reset clears the failsafe total'); + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/t/004_visibility_map_stats.pl b/contrib/vacuum_stats/t/004_visibility_map_stats.pl new file mode 100644 index 00000000000..66256234b09 --- /dev/null +++ b/contrib/vacuum_stats/t/004_visibility_map_stats.pl @@ -0,0 +1,216 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# VM flag clearings are ordinary relation/database statistics. This test +# reads them through the vacuum_stats extension without changing core views. +# Adapt the v44 VM stability scenario to the collector used by Cloudberry: +# https://www.postgresql.org/message-id/attachment/204710/v44-0009-Track-table-VM-stability.patch +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('vm_stats'); +$node->init; +$node->append_conf('postgresql.conf', q{ +autovacuum = off +track_counts = on +}); +$node->start; +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats'); + +sub wait_for_stats +{ + my ($sql, $description) = @_; + $node->poll_query_until('postgres', $sql) + or BAIL_OUT("timed out waiting for $description"); +} + +my $columns = 'frozen_page_marks_cleared, visible_page_marks_cleared'; +my $table_query = "SELECT $columns FROM pg_stat_vacuum_tables WHERE relname = 'vm_heap'"; +my $visible_query = "SELECT $columns FROM pg_stat_vacuum_tables WHERE relname = 'vm_visible'"; +my $db_query = "SELECT $columns FROM pg_stat_vacuum_database WHERE datname = current_database()"; + +sub populated_pages +{ + my ($table) = @_; + return $node->safe_psql('postgres', + "SELECT count(DISTINCT split_part(ctid::text, ',', 1)) FROM $table"); +} + +$node->safe_psql('postgres', q{ +CREATE TABLE vm_heap (id int PRIMARY KEY) WITH (fillfactor = 70); +INSERT INTO vm_heap SELECT generate_series(1, 1000); +CREATE TABLE vm_visible (id int); +INSERT INTO vm_visible SELECT generate_series(1, 1000); +}); +wait_for_stats(q{ +SELECT n_tup_ins = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'insert report'); +$node->safe_psql('postgres', 'VACUUM FREEZE vm_heap'); +wait_for_stats(q{ +SELECT vacuum_count = 1 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'vacuum report'); +# A normal vacuum marks the new tuples visible, without freezing their XIDs. +$node->safe_psql('postgres', q{ +SET vacuum_freeze_min_age = 1000000000; +SET vacuum_freeze_table_age = 1000000000; +VACUUM vm_visible; +}); +wait_for_stats(q{ +SELECT vacuum_count = 1 AND n_tup_ins = 1000 +FROM pg_stat_all_tables_internal WHERE relname = 'vm_visible' +}, 'non-freezing vacuum and insert reports'); + +# Drain reports from table/index creation before taking the database baseline. +$node->restart; +is($node->safe_psql('postgres', $table_query), '0|0', + 'setting VM flags does not count as clearing them'); +my ($db_frozen, $db_visible) = split /\|/, + $node->safe_psql('postgres', $db_query); +my $pages = populated_pages('vm_heap'); +cmp_ok($pages, '>', 0, 'test table has populated heap pages'); + +# Count each physical transition once, even though UPDATE touches many tuples +# on each page. DML and VM counters travel in the same collector message, so +# wait for the DML report before checking exact values, including zeroes. +$node->safe_psql('postgres', 'UPDATE vm_heap SET id = id + 1000'); +wait_for_stats(q{ +SELECT n_tup_upd = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'first update report'); +is($node->safe_psql('postgres', $table_query), "$pages|$pages", + 'UPDATE clears one all-visible and all-frozen mark per populated page'); +is($node->safe_psql('postgres', qq{ +SELECT frozen_page_marks_cleared - $db_frozen, + visible_page_marks_cleared - $db_visible +FROM pg_stat_vacuum_database WHERE datname = current_database() +}), "$pages|$pages", 'database totals include exactly the table transitions'); + +$node->safe_psql('postgres', 'UPDATE vm_heap SET id = id + 1000'); +wait_for_stats(q{ +SELECT n_tup_upd = 2000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'second update report'); +is($node->safe_psql('postgres', $table_query), "$pages|$pages", + 'updates on pages whose marks are already clear add nothing'); +is($node->safe_psql('postgres', qq{ +SELECT pg_stat_get_frozen_page_marks_cleared('vm_heap'::regclass), + pg_stat_get_visible_page_marks_cleared('vm_heap'::regclass) +}), "$pages|$pages", 'extension getters and table view expose the same counters'); + +# Distinguish the two counters and verify that rolling back DML does not undo +# the physical clearing of VM bits. Utility-mode SELECT FOR UPDATE takes a +# table lock, so use unfrozen tuples to exercise independent bit accounting. +my $visible_pages = populated_pages('vm_visible'); +cmp_ok($visible_pages, '>', 0, 'unfrozen table has populated heap pages'); +$node->safe_psql('postgres', 'BEGIN; DELETE FROM vm_visible; ROLLBACK;'); +wait_for_stats(q{ +SELECT n_tup_del = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_visible' +}, 'aborted delete report'); +is($node->safe_psql('postgres', 'SELECT count(*) FROM vm_visible'), '1000', + 'aborted DELETE preserves the rows'); +is($node->safe_psql('postgres', $visible_query), "0|$visible_pages", + 'aborted DELETE counts all-visible clearings without inventing all-frozen clearings'); +is($node->safe_psql('postgres', qq{ +SELECT frozen_page_marks_cleared - $db_frozen, + visible_page_marks_cleared - $db_visible +FROM pg_stat_vacuum_database WHERE datname = current_database() +}), $pages . '|' . ($pages + $visible_pages), + 'database totals distinguish the bits and include aborted DML'); + +# Re-establish flags, then modify the table while a reader holds both an MVCC +# snapshot and a statistics snapshot. The statistics snapshot postpones +# observing the clearings; the VM bits themselves are cleared by DELETE, +# before the reader commits. +$node->safe_psql('postgres', 'VACUUM FREEZE vm_heap'); +wait_for_stats(q{ +SELECT vacuum_count = 2 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'second vacuum report'); +is($node->safe_psql('postgres', $table_query), "$pages|$pages", + 'restoring VM flags does not increase clearing counters'); +my $refrozen_pages = populated_pages('vm_heap'); +my $cleared = $pages + $refrozen_pages; + +my ($in, $out) = ('', ''); +my $timer = IPC::Run::timeout($TestLib::timeout_default); +my $reader = $node->background_psql('postgres', \$in, \$out, $timer); +my $reader_query = sub { + my ($sql) = @_; + $out = ''; + $in = "$sql;\n\\echo vm_query_done\n"; + pump_until($reader, $timer, \$out, qr/^vm_query_done\r?$/m) + or BAIL_OUT('reader did not complete its query'); + $out =~ s/\r//g; + $out =~ s/^vm_query_done\n?//m; + $out =~ s/^\n+|\n+$//g; + return $out; +}; +is($reader_query->("BEGIN ISOLATION LEVEL REPEATABLE READ; $table_query"), + "$pages|$pages", 'reader caches the pre-delete statistics'); +is($reader_query->('SELECT count(*) FROM vm_heap'), '1000', + 'reader holds an MVCC snapshot of the rows'); + +$node->safe_psql('postgres', 'DELETE FROM vm_heap'); +wait_for_stats(q{ +SELECT n_tup_del = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'concurrent delete report'); +is($node->safe_psql('postgres', $table_query), "$cleared|$cleared", + 'DELETE counts fresh VM clearings while the reader transaction is open'); +is($reader_query->($table_query), "$pages|$pages", + 'cached statistics retain their earlier values'); +is($reader_query->("SELECT pg_stat_clear_snapshot(); $table_query"), + "$cleared|$cleared", 'clearing the statistics snapshot exposes the new counters'); +is($reader_query->('SELECT count(*) FROM vm_heap'), '1000', + 'refreshing statistics leaves the MVCC snapshot unchanged'); +$in = "COMMIT;\n\\q\n"; +$reader->finish; +is($node->safe_psql('postgres', $table_query), "$cleared|$cleared", + 'committing the reader adds no clearings'); +is($node->safe_psql('postgres', qq{ +SELECT frozen_page_marks_cleared - $db_frozen, + visible_page_marks_cleared - $db_visible +FROM pg_stat_vacuum_database WHERE datname = current_database() +}), $cleared . '|' . ($cleared + $visible_pages), + 'database totals accumulate both rounds of heap clearings'); + +my $db_totals = $node->safe_psql('postgres', $db_query); +$node->restart; +is($node->safe_psql('postgres', $table_query), "$cleared|$cleared", + 'table VM counters survive a clean restart'); +is($node->safe_psql('postgres', $visible_query), "0|$visible_pages", + 'counts from aborted DML survive a clean restart'); +is($node->safe_psql('postgres', $db_query), $db_totals, + 'database VM counters survive a clean restart'); + +$node->safe_psql('postgres', + q{SELECT pg_stat_reset_single_table_counters('vm_heap'::regclass)}); +wait_for_stats(q{ +SELECT frozen_page_marks_cleared = 0 AND visible_page_marks_cleared = 0 +FROM pg_stat_vacuum_tables WHERE relname = 'vm_heap' +}, 'relation reset'); +is($node->safe_psql('postgres', $table_query), '0|0', + 'relation reset clears both VM counters'); +is($node->safe_psql('postgres', $visible_query), "0|$visible_pages", + 'relation reset preserves another table\'s counters'); +is($node->safe_psql('postgres', $db_query), $db_totals, + 'relation reset preserves database totals'); + +$node->safe_psql('postgres', 'SELECT pg_stat_reset()'); +wait_for_stats(q{ +SELECT frozen_page_marks_cleared = 0 AND visible_page_marks_cleared = 0 +FROM pg_stat_vacuum_database WHERE datname = current_database() +}, 'database reset'); +is($node->safe_psql('postgres', $db_query), '0|0', + 'database reset clears both VM counters'); +is($node->safe_psql('postgres', $visible_query), '0|0', + 'database reset also clears relation VM counters'); + +# The SQL interface belongs to the extension, not the system catalog. +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM pg_attribute +WHERE attrelid IN ('pg_stat_all_tables'::regclass, 'pg_stat_database'::regclass) + AND attname IN ('frozen_page_marks_cleared', 'visible_page_marks_cleared') + AND NOT attisdropped +}), '0', 'core statistics views retain their original columns'); + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/t/005_vacuum_interrupts.pl b/contrib/vacuum_stats/t/005_vacuum_interrupts.pl new file mode 100644 index 00000000000..57967fc491a --- /dev/null +++ b/contrib/vacuum_stats/t/005_vacuum_interrupts.pl @@ -0,0 +1,133 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Adapt v44-0004 to the PG14 collector: wait for heap processing before +# canceling VACUUM, then wait for its deferred statistics report. +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('vacuum_interrupts'); +$node->init; +$node->append_conf('postgresql.conf', "autovacuum = off\n"); +$node->start; +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats'); + +my $db_count = q{SELECT vacuum_interrupt_count FROM pg_stat_vacuum_database + WHERE datname = current_database()}; +my $shared_count = q{SELECT vacuum_interrupt_count FROM pg_stat_vacuum_database + WHERE datid = 0}; +my $nrows = 1000 * ($node->safe_psql('postgres', 'SHOW block_size') / 8192); +$node->safe_psql('postgres', qq{ +CREATE TABLE vacstat_int (id int PRIMARY KEY) + WITH (autovacuum_enabled = off, fillfactor = 10); +INSERT INTO vacstat_int SELECT generate_series(1, $nrows); +DELETE FROM vacstat_int WHERE id % 2 = 0; +}); + +# VERBOSE emits lower-severity messages while the vacuum error callback is +# installed. They must not be mistaken for an interrupted vacuum. +$node->safe_psql('postgres', 'VACUUM (VERBOSE, INDEX_CLEANUP ON) vacstat_int'); +$node->poll_query_until('postgres', q{ +SELECT vacuum_count = 1 FROM pg_stat_all_tables_internal WHERE relname = 'vacstat_int' +}) or BAIL_OUT('successful vacuum report did not reach the collector'); +is($node->safe_psql('postgres', $db_count), '0', + 'successful vacuum and its VERBOSE messages do not count as errors'); + +my ($stdout, $stderr) = ('', ''); +is($node->psql('postgres', 'SELECT 1 / 0', + stdout => \$stdout, stderr => \$stderr), 3, + 'unrelated statement raises an error'); +is($node->safe_psql('postgres', $db_count), '0', + 'an error outside VACUUM does not increment the counter'); + +sub cancel_vacuum +{ + my ($relation, $track_counts) = @_; + my ($in, $out) = ('', ''); + my $timer = IPC::Run::timeout($TestLib::timeout_default); + my $vac = $node->background_psql('postgres', \$in, \$out, $timer, + on_error_stop => 0); + $out = ''; + $in = qq{ +SET application_name = 'vacuum_interrupt_test'; +SET track_counts = $track_counts; +SET vacuum_cost_delay = '100ms'; +SET vacuum_cost_limit = 1; +\\echo vacuum_started +VACUUM (DISABLE_PAGE_SKIPPING) $relation; +\\echo vacuum_done :ERROR :SQLSTATE +}; + pump_until($vac, $timer, \$out, qr/^vacuum_started\r?$/m) + or BAIL_OUT('background psql did not launch VACUUM'); + + # An active VACUUM query alone is insufficient: it might still be waiting + # for a relation lock, before the heap error callback has been installed. + $node->poll_query_until('postgres', qq{ +SELECT count(*) = 1 +FROM pg_stat_activity a JOIN pg_stat_progress_vacuum v USING (pid) +WHERE a.application_name = 'vacuum_interrupt_test' + AND v.relid = '$relation'::regclass AND v.phase = 'scanning heap' + AND a.wait_event = 'VacuumDelay' +}) or BAIL_OUT("VACUUM of $relation did not enter heap processing"); + is($node->safe_psql('postgres', q{ +SELECT pg_cancel_backend(pid) FROM pg_stat_activity +WHERE application_name = 'vacuum_interrupt_test' +}), 't', "sent cancellation to VACUUM of $relation"); + pump_until($vac, $timer, \$out, qr/^vacuum_done\b/m) + or BAIL_OUT('canceled VACUUM did not return'); + like($out, qr/^vacuum_done true 57014\r?$/m, + "VACUUM of $relation reports query_canceled"); + $in = "\\q\n"; + $vac->finish; +} + +for my $expected (1..2) +{ + cancel_vacuum('vacstat_int', 'on'); + ok($node->poll_query_until('postgres', + "SELECT ($db_count) = $expected"), + "canceled vacuum increments database total to exactly $expected"); +} +is($node->safe_psql('postgres', $shared_count), '0', + 'local relation errors leave shared-object statistics alone'); + +cancel_vacuum('vacstat_int', 'off'); +# Drain reports, including the canceled backend's final message, before +# asserting that a disabled counter did not change. +$node->restart; +is($node->safe_psql('postgres', $db_count), '2', + 'track_counts off suppresses counting; earlier errors survive restart'); + +# Make pg_authid large enough to observe cost-delay waits even with 32kB +# blocks. Shared relations report errors to datid zero, not the session's DB. +$node->safe_psql('postgres', qq{ +DO \$\$ BEGIN + FOR i IN 1..$nrows LOOP + EXECUTE format('CREATE ROLE vacstat_role_%s', i); + END LOOP; +END \$\$; +}); +cancel_vacuum('pg_authid', 'on'); +ok($node->poll_query_until('postgres', "SELECT ($shared_count) = 1"), + 'shared relation error reaches the datid zero entry'); +is($node->safe_psql('postgres', $db_count), '2', + 'shared relation error leaves the current database total unchanged'); + +$node->safe_psql('postgres', 'SELECT pg_stat_reset()'); +ok($node->poll_query_until('postgres', "SELECT ($db_count) = 0"), + 'ordinary database reset clears vacuum interruptions'); +$node->restart; +is($node->safe_psql('postgres', $db_count), '0', + 'reset value survives restart'); +is($node->safe_psql('postgres', $shared_count), '1', + 'database reset preserves shared errors, which survive restart'); +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM pg_attribute +WHERE attrelid = 'pg_stat_vacuum_database'::regclass + AND attname = 'vacuum_interrupt_count' AND NOT attisdropped +}), '1', 'extension view exposes vacuum_interrupt_count'); + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/vacuum_stats--1.0.sql b/contrib/vacuum_stats/vacuum_stats--1.0.sql new file mode 100644 index 00000000000..5cf0b4926c1 --- /dev/null +++ b/contrib/vacuum_stats/vacuum_stats--1.0.sql @@ -0,0 +1,558 @@ +/* contrib/vacuum_stats/vacuum_stats--1.0.sql */ + +-- complain if script is sourced in psql, rather than via CREATE EXTENSION +\echo Use "CREATE EXTENSION vacuum_stats" to load this file. \quit + +-- +-- Per-relation accessor functions (tables and indexes). +-- +CREATE FUNCTION pg_stat_get_vacuum_tuples_deleted(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_tuples_deleted' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_dead_tuples(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_dead_tuples' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_pages_deleted(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_pages_deleted' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_bytes_removed(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_bytes_removed' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_total_file_segs(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_total_file_segs' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_dead_pages(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_dead_pages' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_pages_frozen(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_pages_frozen' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_pages_all_visible(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_pages_all_visible' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_freeze_age_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_freeze_age_count' +LANGUAGE C STABLE STRICT; + +-- +-- Per-database accessor functions. +-- +CREATE FUNCTION pg_stat_get_db_vacuum_tuples_deleted(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_tuples_deleted' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_dead_tuples(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_dead_tuples' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_pages_deleted(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_pages_deleted' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_bytes_removed(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_bytes_removed' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_dead_pages(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_dead_pages' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_pages_frozen(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_pages_frozen' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_pages_all_visible(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_pages_all_visible' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_freeze_age_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_freeze_age_count' +LANGUAGE C STABLE STRICT; + +-- Access to counters stored in ordinary relation and database statistics. +CREATE FUNCTION pg_stat_get_frozen_page_marks_cleared(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_frozen_page_marks_cleared' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_visible_page_marks_cleared(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_visible_page_marks_cleared' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_vacuum_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_vacuum_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_autovacuum_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_autovacuum_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_vacuum_delay_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_vacuum_delay_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_autovacuum_delay_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_autovacuum_delay_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_analyze_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_analyze_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_autoanalyze_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_autoanalyze_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_vacuum_failsafe_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_failsafe_count' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_frozen_page_marks_cleared(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_frozen_page_marks_cleared' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_visible_page_marks_cleared(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_visible_page_marks_cleared' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_total_vacuum_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_vacuum_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_total_autovacuum_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_autovacuum_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_total_vacuum_delay_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_vacuum_delay_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_total_autovacuum_delay_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_autovacuum_delay_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_vacuum_failsafe_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_failsafe_count' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_vacuum_interrupt_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_interrupt_count' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +-- Local statistics, in milliseconds for timing columns. +CREATE VIEW pg_stat_vacuum_tables AS + SELECT + c.oid AS relid, + n.nspname AS schemaname, + c.relname AS relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(c.oid) AS tuples_deleted, + @extschema@.pg_stat_get_vacuum_dead_tuples(c.oid) AS dead_tuples, + @extschema@.pg_stat_get_vacuum_pages_deleted(c.oid) AS pages_deleted, + @extschema@.pg_stat_get_vacuum_bytes_removed(c.oid) AS bytes_removed, + @extschema@.pg_stat_get_vacuum_dead_pages(c.oid) AS dead_pages, + @extschema@.pg_stat_get_vacuum_pages_frozen(c.oid) AS pages_frozen, + @extschema@.pg_stat_get_vacuum_pages_all_visible(c.oid) AS pages_all_visible, + @extschema@.pg_stat_get_frozen_page_marks_cleared(c.oid) AS frozen_page_marks_cleared, + @extschema@.pg_stat_get_visible_page_marks_cleared(c.oid) AS visible_page_marks_cleared, + @extschema@.pg_stat_get_vacuum_freeze_age_count(c.oid) AS freeze_age_vacuum_count, + @extschema@.pg_stat_get_vacuum_failsafe_count(c.oid) AS vacuum_failsafe_count, + @extschema@.pg_stat_get_total_vacuum_time(c.oid) AS total_vacuum_time, + @extschema@.pg_stat_get_total_autovacuum_time(c.oid) AS total_autovacuum_time, + @extschema@.pg_stat_get_total_vacuum_delay_time(c.oid) AS total_vacuum_delay_time, + @extschema@.pg_stat_get_total_autovacuum_delay_time(c.oid) AS total_autovacuum_delay_time, + @extschema@.pg_stat_get_total_analyze_time(c.oid) AS total_analyze_time, + @extschema@.pg_stat_get_total_autoanalyze_time(c.oid) AS total_autoanalyze_time, + @extschema@.pg_stat_get_vacuum_total_file_segs(c.oid) AS total_file_segs + FROM pg_catalog.pg_class c + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm'); + +CREATE VIEW pg_stat_vacuum_indexes AS + SELECT + c.oid AS relid, + i.oid AS indexrelid, + n.nspname AS schemaname, + c.relname AS relname, + i.relname AS indexrelname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(i.oid) AS tuples_deleted, + @extschema@.pg_stat_get_vacuum_dead_tuples(i.oid) AS dead_tuples, + @extschema@.pg_stat_get_vacuum_pages_deleted(i.oid) AS pages_deleted, + @extschema@.pg_stat_get_vacuum_bytes_removed(i.oid) AS bytes_removed, + @extschema@.pg_stat_get_vacuum_dead_pages(i.oid) AS dead_pages, + @extschema@.pg_stat_get_vacuum_pages_frozen(i.oid) AS pages_frozen, + @extschema@.pg_stat_get_vacuum_pages_all_visible(i.oid) AS pages_all_visible, + @extschema@.pg_stat_get_frozen_page_marks_cleared(i.oid) AS frozen_page_marks_cleared, + @extschema@.pg_stat_get_visible_page_marks_cleared(i.oid) AS visible_page_marks_cleared, + @extschema@.pg_stat_get_vacuum_freeze_age_count(i.oid) AS freeze_age_vacuum_count, + @extschema@.pg_stat_get_vacuum_failsafe_count(i.oid) AS vacuum_failsafe_count, + @extschema@.pg_stat_get_total_vacuum_time(i.oid) AS total_vacuum_time, + @extschema@.pg_stat_get_total_autovacuum_time(i.oid) AS total_autovacuum_time, + @extschema@.pg_stat_get_total_vacuum_delay_time(i.oid) AS total_vacuum_delay_time, + @extschema@.pg_stat_get_total_autovacuum_delay_time(i.oid) AS total_autovacuum_delay_time + FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_index x ON c.oid = x.indrelid + JOIN pg_catalog.pg_class i ON i.oid = x.indexrelid + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') AND i.relkind = 'i'; + +CREATE VIEW pg_stat_vacuum_database AS + SELECT + d.oid AS datid, + d.datname AS datname, + @extschema@.pg_stat_get_db_vacuum_tuples_deleted(d.oid) AS tuples_deleted, + @extschema@.pg_stat_get_db_vacuum_dead_tuples(d.oid) AS dead_tuples, + @extschema@.pg_stat_get_db_vacuum_pages_deleted(d.oid) AS pages_deleted, + @extschema@.pg_stat_get_db_vacuum_bytes_removed(d.oid) AS bytes_removed, + @extschema@.pg_stat_get_db_vacuum_dead_pages(d.oid) AS dead_pages, + @extschema@.pg_stat_get_db_vacuum_pages_frozen(d.oid) AS pages_frozen, + @extschema@.pg_stat_get_db_vacuum_pages_all_visible(d.oid) AS pages_all_visible, + @extschema@.pg_stat_get_db_frozen_page_marks_cleared(d.oid) AS frozen_page_marks_cleared, + @extschema@.pg_stat_get_db_visible_page_marks_cleared(d.oid) AS visible_page_marks_cleared, + @extschema@.pg_stat_get_db_vacuum_freeze_age_count(d.oid) AS freeze_age_vacuum_count, + @extschema@.pg_stat_get_db_vacuum_failsafe_count(d.oid) AS vacuum_failsafe_count, + @extschema@.pg_stat_get_db_total_vacuum_time(d.oid) AS total_vacuum_time, + @extschema@.pg_stat_get_db_total_autovacuum_time(d.oid) AS total_autovacuum_time, + @extschema@.pg_stat_get_db_total_vacuum_delay_time(d.oid) AS total_vacuum_delay_time, + @extschema@.pg_stat_get_db_total_autovacuum_delay_time(d.oid) AS total_autovacuum_delay_time, + @extschema@.pg_stat_get_db_vacuum_interrupt_count(d.oid) AS vacuum_interrupt_count + FROM ( + SELECT 0::oid AS oid, NULL::name AS datname + UNION ALL + SELECT oid, datname FROM pg_catalog.pg_database + ) d; + +-- Cluster views execute on the coordinator and all segments. The bodies +-- access catalogs directly because segment functions cannot scan local views. +-- Utility sessions return their local rows once, without a segment branch. +CREATE FUNCTION gp_stat_get_coordinator_vacuum_tables() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + c.oid, + n.nspname, + c.relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(c.oid), + @extschema@.pg_stat_get_vacuum_dead_tuples(c.oid), + @extschema@.pg_stat_get_vacuum_pages_deleted(c.oid), + @extschema@.pg_stat_get_vacuum_bytes_removed(c.oid), + @extschema@.pg_stat_get_vacuum_dead_pages(c.oid), + @extschema@.pg_stat_get_vacuum_pages_frozen(c.oid), + @extschema@.pg_stat_get_vacuum_pages_all_visible(c.oid), + @extschema@.pg_stat_get_frozen_page_marks_cleared(c.oid), + @extschema@.pg_stat_get_visible_page_marks_cleared(c.oid), + @extschema@.pg_stat_get_vacuum_freeze_age_count(c.oid), + @extschema@.pg_stat_get_vacuum_failsafe_count(c.oid), + @extschema@.pg_stat_get_total_vacuum_time(c.oid), + @extschema@.pg_stat_get_total_autovacuum_time(c.oid), + @extschema@.pg_stat_get_total_vacuum_delay_time(c.oid), + @extschema@.pg_stat_get_total_autovacuum_delay_time(c.oid), + @extschema@.pg_stat_get_total_analyze_time(c.oid), + @extschema@.pg_stat_get_total_autoanalyze_time(c.oid), + @extschema@.pg_stat_get_vacuum_total_file_segs(c.oid) + FROM pg_catalog.pg_class c + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') +$$ +LANGUAGE SQL EXECUTE ON COORDINATOR; + +CREATE FUNCTION gp_stat_get_segment_vacuum_tables() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + c.oid, + n.nspname, + c.relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(c.oid), + @extschema@.pg_stat_get_vacuum_dead_tuples(c.oid), + @extschema@.pg_stat_get_vacuum_pages_deleted(c.oid), + @extschema@.pg_stat_get_vacuum_bytes_removed(c.oid), + @extschema@.pg_stat_get_vacuum_dead_pages(c.oid), + @extschema@.pg_stat_get_vacuum_pages_frozen(c.oid), + @extschema@.pg_stat_get_vacuum_pages_all_visible(c.oid), + @extschema@.pg_stat_get_frozen_page_marks_cleared(c.oid), + @extschema@.pg_stat_get_visible_page_marks_cleared(c.oid), + @extschema@.pg_stat_get_vacuum_freeze_age_count(c.oid), + @extschema@.pg_stat_get_vacuum_failsafe_count(c.oid), + @extschema@.pg_stat_get_total_vacuum_time(c.oid), + @extschema@.pg_stat_get_total_autovacuum_time(c.oid), + @extschema@.pg_stat_get_total_vacuum_delay_time(c.oid), + @extschema@.pg_stat_get_total_autovacuum_delay_time(c.oid), + @extschema@.pg_stat_get_total_analyze_time(c.oid), + @extschema@.pg_stat_get_total_autoanalyze_time(c.oid), + @extschema@.pg_stat_get_vacuum_total_file_segs(c.oid) + FROM pg_catalog.pg_class c + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') + AND pg_catalog.current_setting('gp_role') <> 'utility' +$$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +CREATE VIEW gp_stat_vacuum_tables AS + SELECT * FROM gp_stat_get_coordinator_vacuum_tables() AS T + (gp_segment_id int, + relid oid, + schemaname name, + relname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision, + total_analyze_time double precision, + total_autoanalyze_time double precision, + total_file_segs int8) + UNION ALL + SELECT * FROM gp_stat_get_segment_vacuum_tables() AS T + (gp_segment_id int, + relid oid, + schemaname name, + relname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision, + total_analyze_time double precision, + total_autoanalyze_time double precision, + total_file_segs int8); + +CREATE FUNCTION gp_stat_get_coordinator_vacuum_indexes() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + c.oid, + i.oid, + n.nspname, + c.relname, + i.relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(i.oid), + @extschema@.pg_stat_get_vacuum_dead_tuples(i.oid), + @extschema@.pg_stat_get_vacuum_pages_deleted(i.oid), + @extschema@.pg_stat_get_vacuum_bytes_removed(i.oid), + @extschema@.pg_stat_get_vacuum_dead_pages(i.oid), + @extschema@.pg_stat_get_vacuum_pages_frozen(i.oid), + @extschema@.pg_stat_get_vacuum_pages_all_visible(i.oid), + @extschema@.pg_stat_get_frozen_page_marks_cleared(i.oid), + @extschema@.pg_stat_get_visible_page_marks_cleared(i.oid), + @extschema@.pg_stat_get_vacuum_freeze_age_count(i.oid), + @extschema@.pg_stat_get_vacuum_failsafe_count(i.oid), + @extschema@.pg_stat_get_total_vacuum_time(i.oid), + @extschema@.pg_stat_get_total_autovacuum_time(i.oid), + @extschema@.pg_stat_get_total_vacuum_delay_time(i.oid), + @extschema@.pg_stat_get_total_autovacuum_delay_time(i.oid) + FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_index x ON c.oid = x.indrelid + JOIN pg_catalog.pg_class i ON i.oid = x.indexrelid + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') AND i.relkind = 'i' +$$ +LANGUAGE SQL EXECUTE ON COORDINATOR; + +CREATE FUNCTION gp_stat_get_segment_vacuum_indexes() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + c.oid, + i.oid, + n.nspname, + c.relname, + i.relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(i.oid), + @extschema@.pg_stat_get_vacuum_dead_tuples(i.oid), + @extschema@.pg_stat_get_vacuum_pages_deleted(i.oid), + @extschema@.pg_stat_get_vacuum_bytes_removed(i.oid), + @extschema@.pg_stat_get_vacuum_dead_pages(i.oid), + @extschema@.pg_stat_get_vacuum_pages_frozen(i.oid), + @extschema@.pg_stat_get_vacuum_pages_all_visible(i.oid), + @extschema@.pg_stat_get_frozen_page_marks_cleared(i.oid), + @extschema@.pg_stat_get_visible_page_marks_cleared(i.oid), + @extschema@.pg_stat_get_vacuum_freeze_age_count(i.oid), + @extschema@.pg_stat_get_vacuum_failsafe_count(i.oid), + @extschema@.pg_stat_get_total_vacuum_time(i.oid), + @extschema@.pg_stat_get_total_autovacuum_time(i.oid), + @extschema@.pg_stat_get_total_vacuum_delay_time(i.oid), + @extschema@.pg_stat_get_total_autovacuum_delay_time(i.oid) + FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_index x ON c.oid = x.indrelid + JOIN pg_catalog.pg_class i ON i.oid = x.indexrelid + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') AND i.relkind = 'i' + AND pg_catalog.current_setting('gp_role') <> 'utility' +$$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +CREATE VIEW gp_stat_vacuum_indexes AS + SELECT * FROM gp_stat_get_coordinator_vacuum_indexes() AS T + (gp_segment_id int, + relid oid, + indexrelid oid, + schemaname name, + relname name, + indexrelname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision) + UNION ALL + SELECT * FROM gp_stat_get_segment_vacuum_indexes() AS T + (gp_segment_id int, + relid oid, + indexrelid oid, + schemaname name, + relname name, + indexrelname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision); + +CREATE FUNCTION gp_stat_get_coordinator_vacuum_database() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + d.oid, + d.datname, + @extschema@.pg_stat_get_db_vacuum_tuples_deleted(d.oid), + @extschema@.pg_stat_get_db_vacuum_dead_tuples(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_deleted(d.oid), + @extschema@.pg_stat_get_db_vacuum_bytes_removed(d.oid), + @extschema@.pg_stat_get_db_vacuum_dead_pages(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_frozen(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_all_visible(d.oid), + @extschema@.pg_stat_get_db_frozen_page_marks_cleared(d.oid), + @extschema@.pg_stat_get_db_visible_page_marks_cleared(d.oid), + @extschema@.pg_stat_get_db_vacuum_freeze_age_count(d.oid), + @extschema@.pg_stat_get_db_vacuum_failsafe_count(d.oid), + @extschema@.pg_stat_get_db_total_vacuum_time(d.oid), + @extschema@.pg_stat_get_db_total_autovacuum_time(d.oid), + @extschema@.pg_stat_get_db_total_vacuum_delay_time(d.oid), + @extschema@.pg_stat_get_db_total_autovacuum_delay_time(d.oid), + @extschema@.pg_stat_get_db_vacuum_interrupt_count(d.oid) + FROM ( + SELECT 0::oid AS oid, NULL::name AS datname + UNION ALL + SELECT oid, datname FROM pg_catalog.pg_database + ) d +$$ +LANGUAGE SQL EXECUTE ON COORDINATOR; + +CREATE FUNCTION gp_stat_get_segment_vacuum_database() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + d.oid, + d.datname, + @extschema@.pg_stat_get_db_vacuum_tuples_deleted(d.oid), + @extschema@.pg_stat_get_db_vacuum_dead_tuples(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_deleted(d.oid), + @extschema@.pg_stat_get_db_vacuum_bytes_removed(d.oid), + @extschema@.pg_stat_get_db_vacuum_dead_pages(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_frozen(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_all_visible(d.oid), + @extschema@.pg_stat_get_db_frozen_page_marks_cleared(d.oid), + @extschema@.pg_stat_get_db_visible_page_marks_cleared(d.oid), + @extschema@.pg_stat_get_db_vacuum_freeze_age_count(d.oid), + @extschema@.pg_stat_get_db_vacuum_failsafe_count(d.oid), + @extschema@.pg_stat_get_db_total_vacuum_time(d.oid), + @extschema@.pg_stat_get_db_total_autovacuum_time(d.oid), + @extschema@.pg_stat_get_db_total_vacuum_delay_time(d.oid), + @extschema@.pg_stat_get_db_total_autovacuum_delay_time(d.oid), + @extschema@.pg_stat_get_db_vacuum_interrupt_count(d.oid) + FROM ( + SELECT 0::oid AS oid, NULL::name AS datname + UNION ALL + SELECT oid, datname FROM pg_catalog.pg_database + ) d + WHERE pg_catalog.current_setting('gp_role') <> 'utility' +$$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +CREATE VIEW gp_stat_vacuum_database AS + SELECT * FROM gp_stat_get_coordinator_vacuum_database() AS T + (gp_segment_id int, + datid oid, + datname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision, + vacuum_interrupt_count int8) + UNION ALL + SELECT * FROM gp_stat_get_segment_vacuum_database() AS T + (gp_segment_id int, + datid oid, + datname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision, + vacuum_interrupt_count int8); + +GRANT SELECT ON pg_stat_vacuum_tables, pg_stat_vacuum_indexes, + pg_stat_vacuum_database, gp_stat_vacuum_tables, + gp_stat_vacuum_indexes, gp_stat_vacuum_database TO PUBLIC; diff --git a/contrib/vacuum_stats/vacuum_stats.c b/contrib/vacuum_stats/vacuum_stats.c new file mode 100644 index 00000000000..a84a563b612 --- /dev/null +++ b/contrib/vacuum_stats/vacuum_stats.c @@ -0,0 +1,160 @@ +/*------------------------------------------------------------------------- + * + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + * + * vacuum_stats.c + * Expose the vacuum counters accumulated by the statistics collector + * for relations (tables and indexes) and databases. + * + * The backend collects these counters in ordinary statistics entries and, + * when enabled, extended vacuum statistics. SQL access belongs to this + * extension; read both groups through the regular pgstat fetch API. + * + * contrib/vacuum_stats/vacuum_stats.c + * + *------------------------------------------------------------------------- + */ +#include "postgres.h" + +#include "fmgr.h" +#include "pgstat.h" + +PG_MODULE_MAGIC; + +/* Ordinary counters follow track_counts, independently of extended tracking. */ +#define DEFINE_REL_COUNTER_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + PgStat_StatTabEntry *entry = pgstat_fetch_stat_tabentry(PG_GETARG_OID(0)); \ + PG_RETURN_INT64(entry ? (int64) entry->field : 0); \ +} + +#define DEFINE_DB_COUNTER_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + PgStat_StatDBEntry *entry = pgstat_fetch_stat_dbentry(PG_GETARG_OID(0)); \ + PG_RETURN_INT64(entry ? (int64) entry->field : 0); \ +} + +/* Store microseconds in the collector and expose milliseconds without rounding. */ +#define DEFINE_REL_TIME_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + PgStat_StatTabEntry *entry = pgstat_fetch_stat_tabentry(PG_GETARG_OID(0)); \ + PG_RETURN_FLOAT8(entry ? (double) entry->field / 1000.0 : 0); \ +} + +#define DEFINE_DB_TIME_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + PgStat_StatDBEntry *entry = pgstat_fetch_stat_dbentry(PG_GETARG_OID(0)); \ + PG_RETURN_FLOAT8(entry ? (double) entry->field / 1000.0 : 0); \ +} + +DEFINE_REL_COUNTER_FUNC(pg_stat_get_frozen_page_marks_cleared, frozen_page_marks_cleared) +DEFINE_REL_COUNTER_FUNC(pg_stat_get_visible_page_marks_cleared, visible_page_marks_cleared) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_vacuum_time, total_vacuum_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_autovacuum_time, total_autovacuum_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_vacuum_delay_time, total_vacuum_delay_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_autovacuum_delay_time, total_autovacuum_delay_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_analyze_time, total_analyze_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_autoanalyze_time, total_autoanalyze_time) +DEFINE_REL_COUNTER_FUNC(pg_stat_get_vacuum_failsafe_count, vacuum_failsafe_count) +DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_frozen_page_marks_cleared, n_frozen_page_marks_cleared) +DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_visible_page_marks_cleared, n_visible_page_marks_cleared) +DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_vacuum_time, total_vacuum_time) +DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_autovacuum_time, total_autovacuum_time) +DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_vacuum_delay_time, total_vacuum_delay_time) +DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_autovacuum_delay_time, total_autovacuum_delay_time) +DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_vacuum_failsafe_count, vacuum_failsafe_count) +DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_vacuum_interrupt_count, vacuum_interrupt_count) + +/* Fetch the relation's vacuum counters, or NULL when unavailable. */ +static PgStat_VacuumStats * +fetch_rel_vacuum_stats(Oid relid) +{ + return pgstat_fetch_stat_vacuum_stats(relid); +} + +/* + * Fetch the per-database vacuum counters, or NULL if the statistics + * collector has no entry for the database. + */ +static PgStat_VacuumStats * +fetch_db_vacuum_stats(Oid dbid) +{ + PgStat_StatDBEntry *dbentry; + + if (!pgstat_track_vacuum_statistics) + return NULL; + + dbentry = pgstat_fetch_stat_dbentry(dbid); + if (dbentry == NULL) + return NULL; + + return &dbentry->n_vacuum_stats; +} + +#define DEFINE_REL_VACSTAT_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + Oid relid = PG_GETARG_OID(0); \ + PgStat_VacuumStats *stats = fetch_rel_vacuum_stats(relid); \ +\ + PG_RETURN_INT64(stats ? (int64) stats->field : 0); \ +} + +#define DEFINE_DB_VACSTAT_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + Oid dbid = PG_GETARG_OID(0); \ + PgStat_VacuumStats *stats = fetch_db_vacuum_stats(dbid); \ +\ + PG_RETURN_INT64(stats ? (int64) stats->field : 0); \ +} + +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_tuples_deleted, tuples_deleted) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_dead_tuples, dead_tuples) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_pages_deleted, pages_deleted) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_bytes_removed, bytes_removed) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_total_file_segs, total_file_segs) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_dead_pages, dead_pages) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_pages_frozen, pages_frozen) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_pages_all_visible, pages_all_visible) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_freeze_age_count, freeze_age_vacuum_count) + +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_tuples_deleted, tuples_deleted) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_dead_tuples, dead_tuples) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_deleted, pages_deleted) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_bytes_removed, bytes_removed) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_dead_pages, dead_pages) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_frozen, pages_frozen) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_all_visible, pages_all_visible) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_freeze_age_count, freeze_age_vacuum_count) diff --git a/contrib/vacuum_stats/vacuum_stats.control b/contrib/vacuum_stats/vacuum_stats.control new file mode 100644 index 00000000000..5f80449984d --- /dev/null +++ b/contrib/vacuum_stats/vacuum_stats.control @@ -0,0 +1,5 @@ +# vacuum_stats extension +comment = 'per-relation and per-database vacuum statistics' +default_version = '1.0' +module_pathname = '$libdir/vacuum_stats' +relocatable = false From cb934b82cb2a46d8daceea2436e091ce23815db7 Mon Sep 17 00:00:00 2001 From: Alena Rybakina Date: Mon, 21 Sep 2026 16:04:50 +0300 Subject: [PATCH 18/18] Reset vacuum counters for a relation or the current database Add extension functions that reset extended vacuum counters, VM clearings, manual/autovacuum elapsed and cost-delay times, and vacuum_failsafe_count. Preserve ordinary row, access and maintenance counters, timestamps and ANALYZE times. Relation resets leave other relations and database totals intact; database resets exclude shared catalogs. Provide segment dispatch wrappers and restrict execution through SQL grants. Distinguish database-wide resets explicitly in collector messages. Reject OID zero in the relation overload and route shared relations and indexes only to the shared collector entry. NULL relation arguments do nothing. Reset ordinary vacuum fields even when extended tracking is off. Cover reset boundaries, invalid and NULL OIDs, shared catalogs and indexes, AO segment snapshots and database totals in TAP. Synchronize with collector-report wait helpers. Check extension installation in a non-default schema, unchanged system statistics views and built-in functions, shared-relation rows, preservation of ANALYZE times on vacuum reset, and collector statistics across extension removal and reinstallation. --- contrib/vacuum_stats/README.md | 5 + .../vacuum_stats/t/001_vacuum_statistics.pl | 129 ++++++++++++++++++ .../vacuum_stats/t/006_extension_catalog.pl | 90 ++++++++++++ contrib/vacuum_stats/vacuum_stats--1.0.sql | 30 ++++ contrib/vacuum_stats/vacuum_stats.c | 29 ++++ src/backend/postmaster/pgstat.c | 99 ++++++++++++++ src/include/pgstat.h | 17 +++ 7 files changed, 399 insertions(+) create mode 100644 contrib/vacuum_stats/t/006_extension_catalog.pl diff --git a/contrib/vacuum_stats/README.md b/contrib/vacuum_stats/README.md index a0cf2982e80..4f1252dc2ec 100644 --- a/contrib/vacuum_stats/README.md +++ b/contrib/vacuum_stats/README.md @@ -17,6 +17,11 @@ including vacuum times, VM clearings, failsafe and interruption counts. The `visible_page_marks_cleared` and `frozen_page_marks_cleared` counters describe visibility-map changes caused by data modifications. They follow `track_counts` and are collected and retained independently of `track_vacuum_statistics`. +The dedicated vacuum-statistics reset functions also reset these counters, +vacuum times and failsafe counts. Database-wide vacuum reset also +clears `pg_stat_vacuum_database.vacuum_interrupt_count`; relation reset preserves +this database total. The reset functions preserve ordinary access counters, +maintenance counts and timestamps, and ANALYZE times. Install with `CREATE EXTENSION vacuum_stats`. All new SQL functions and views belong to the extension's schema. Existing system views and built-in diff --git a/contrib/vacuum_stats/t/001_vacuum_statistics.pl b/contrib/vacuum_stats/t/001_vacuum_statistics.pl index bf25446f135..8fc4d92558c 100644 --- a/contrib/vacuum_stats/t/001_vacuum_statistics.pl +++ b/contrib/vacuum_stats/t/001_vacuum_statistics.pl @@ -518,6 +518,9 @@ sub snapshot_bytes vacuum_table($table); is(counters($table, 'bytes_removed, pages_deleted, total_file_segs'), "$tail_size|$pages|$segs", "$am repeated vacuum does not recount the tail or segments"); + run_and_wait("SELECT vacuum_stats_reset('$table'::regclass::oid)"); + is(counters($table, 'bytes_removed, total_file_segs'), '0|0', + "$am relation reset clears both cumulative bytes and segment state"); } }; @@ -596,6 +599,87 @@ sub snapshot_bytes 't', 'deleted-page totals accumulate without recounting the first run'); }; +subtest 'reset preserves ordinary statistics and other relations' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_reset (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +CREATE TABLE vstat_keep (id int) WITH (autovacuum_enabled = off); +INSERT INTO vstat_reset SELECT generate_series(1, 1000); +INSERT INTO vstat_keep SELECT generate_series(1, 1000); +DELETE FROM vstat_keep; +}); + vacuum_table('vstat_keep'); + vacuum_table('vstat_reset', 'FREEZE'); + $node->safe_psql('postgres', 'UPDATE vstat_reset SET id = id + 1000'); + wait_for_updates('vstat_reset', 1000); + vacuum_table('vstat_reset', 'FREEZE'); + is(counters('vstat_reset', + 'tuples_deleted, frozen_page_marks_cleared > 0, visible_page_marks_cleared > 0'), + '1000|t|t', 'populate removal and VM revision counters before reset'); + my $keep_before = counters('vstat_keep'); + my $index_before = index_counters('vstat_reset_pkey'); + my $db_before = database_counters(); + my $ordinary_query = "SELECT vacuum_count, n_tup_upd FROM pg_stat_all_tables_internal WHERE relname = 'vstat_reset'"; + my $ordinary_before = $node->safe_psql('postgres', $ordinary_query); + my $reset_before = counters('vstat_reset'); + + my ($result, $stdout, $stderr) = $node->psql('postgres', + 'SELECT vacuum_stats_reset(0::oid)'); + is($result, 3, 'an invalid relation OID is rejected'); + like($stderr, qr/invalid relation OID: 0/, 'invalid OID has a specific error'); + run_and_wait('SELECT vacuum_stats_reset(NULL::oid)'); + is(counters('vstat_reset'), $reset_before, + 'invalid and NULL OIDs do not reset a relation'); + is(database_counters(), $db_before, + 'invalid and NULL OIDs do not reset database totals'); + + run_and_wait("SELECT vacuum_stats_reset('vstat_reset'::regclass::oid)"); + is(counters('vstat_reset', $all_zero), 't', 'relation reset clears every vacuum counter including VM clearing counters'); + is(counters('vstat_keep'), $keep_before, 'relation reset preserves another table'); + is(index_counters('vstat_reset_pkey'), $index_before, 'relation reset preserves the index counters'); + is(database_counters(), $db_before, 'relation reset preserves database totals'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'relation reset preserves ordinary statistics'); + + # Refill the reset relation, including both VM revision counters, + # before checking the database-wide reset. + $node->safe_psql('postgres', 'UPDATE vstat_reset SET id = id + 1000'); + wait_for_updates('vstat_reset', 2000); + vacuum_table('vstat_reset', 'FREEZE'); + is(counters('vstat_reset', + 'tuples_deleted, frozen_page_marks_cleared > 0, visible_page_marks_cleared > 0'), + '1000|t|t', 'the counters are nonzero again before database reset'); + $ordinary_before = $node->safe_psql('postgres', $ordinary_query); + run_and_wait('SELECT vacuum_stats_reset()'); + is($node->safe_psql('postgres', "SELECT bool_and($all_zero) FROM pg_stat_vacuum_tables"), + 't', 'database reset clears all table counters'); + is($node->safe_psql('postgres', 'SELECT bool_and(total_file_segs = 0) FROM pg_stat_vacuum_tables'), + 't', 'database reset clears AO segment snapshots'); + is($node->safe_psql('postgres', "SELECT bool_and($all_zero) FROM pg_stat_vacuum_indexes"), + 't', 'database reset clears all index counters'); + is(database_counters($all_zero), 't', 'database reset clears database totals'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'database reset preserves ordinary statistics'); + + +}; + +subtest 'database totals do not count index work twice' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_db (id int PRIMARY KEY, val int) WITH (autovacuum_enabled = off); +CREATE INDEX ON vstat_db (val); +INSERT INTO vstat_db SELECT i, i FROM generate_series(1, 10000) g(i); +DELETE FROM vstat_db WHERE id % 2 = 0; +}); + run_and_wait('SELECT vacuum_stats_reset()'); + vacuum_table('vstat_db'); + is($node->safe_psql('postgres', + "SELECT sum(tuples_deleted) FROM pg_stat_vacuum_indexes WHERE relname = 'vstat_db'"), + '10000', 'both indexes report 5000 removed entries'); + is(database_counters('tuples_deleted'), '5000', 'database totals count only heap tuples'); + is(database_counters(), counters('vstat_db'), + 'all database vacuum counters match the only table vacuumed since reset'); +}; + subtest 'statistics snapshots release their vacuum counters' => sub { # Keep the last snapshot alive until the context count is read. # Outside this transaction its context would already have been freed. @@ -615,6 +699,41 @@ END }), '1', 'only the current snapshot owns statistics entries with vacuum counters'); }; +subtest 'reset targets shared catalogs independently' => sub { + # Shared catalogs have a separate collector entry, not the current DB's. + # Create actual dead index entries instead of timing an empty cleanup. + $node->safe_psql('postgres', 'CREATE DATABASE vstat_shared_reset'); + $node->safe_psql('postgres', 'DROP DATABASE vstat_shared_reset'); + vacuum_table('pg_database', '(FREEZE, INDEX_CLEANUP ON)'); + is(counters('pg_database', 'total_vacuum_time > 0'), 't', + 'populate vacuum counters for a shared catalog'); + my $shared_before = counters('pg_database'); + run_and_wait('SELECT vacuum_stats_reset()'); + is(counters('pg_database'), $shared_before, + 'database reset leaves shared catalog counters alone'); + + vacuum_table('vstat_reset'); + my $reset_before = counters('vstat_reset'); + my $db_before = database_counters(); + run_and_wait("SELECT vacuum_stats_reset('pg_database'::regclass::oid)"); + is(counters('pg_database', $all_zero), 't', + 'relation reset reaches the shared catalog entry'); + is(counters('vstat_reset'), $reset_before, + 'shared relation reset leaves current-database relations alone'); + is(database_counters(), $db_before, + 'shared relation reset leaves current-database totals alone'); + + # Indexes are separate reset targets, including shared catalog indexes. + is(index_counters('pg_database_oid_index', 'tuples_deleted > 0'), 't', + 'resetting the shared table preserves its index counters'); + run_and_wait(q{ +SELECT vacuum_stats_reset(indexrelid) FROM pg_index +WHERE indrelid = 'pg_database'::regclass +}); + is(index_counters('pg_database_oid_index', $all_zero), 't', + 'relation reset also reaches shared index counters'); +}; + subtest 'disabling tracking omits vacuum counters and preserves ordinary statistics' => sub { $node->safe_psql('postgres', q{ CREATE TABLE vstat_startup (id int PRIMARY KEY) WITH (autovacuum_enabled = off); @@ -653,6 +772,16 @@ END 'reading statistics with tracking disabled preserves ordinary counters'); cmp_ok(snapshot_bytes(), '<', $on_bytes, 'disabled snapshots omit vacuum storage but retain ordinary VM counters'); + run_and_wait("SELECT vacuum_stats_reset('vstat_startup'::regclass::oid)"); + is(counters('vstat_startup', $vm_columns), '0|0', + 'relation reset clears VM counters with vacuum tracking off'); + is(database_counters($vm_columns), $db_vm_before, + 'relation reset preserves database VM totals with vacuum tracking off'); + run_and_wait('SELECT vacuum_stats_reset()'); + is(database_counters($vm_columns), '0|0', + 'database reset clears VM totals with vacuum tracking off'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'vacuum reset while disabled leaves ordinary counters intact'); $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = on'); $node->restart; is(counters('vstat_startup', $vacuum_zero), 't', diff --git a/contrib/vacuum_stats/t/006_extension_catalog.pl b/contrib/vacuum_stats/t/006_extension_catalog.pl new file mode 100644 index 00000000000..900542c17ef --- /dev/null +++ b/contrib/vacuum_stats/t/006_extension_catalog.pl @@ -0,0 +1,90 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# SQL access belongs to an extension, including counters collected while the +# extended vacuum statistics are disabled. Check installation in another schema +# and removal/reinstallation without altering the system statistics views. +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('extension_catalog'); +$node->init; +$node->append_conf('postgresql.conf', "autovacuum = off\n"); +$node->start; + +my $catalog_query = q{ +SELECT c.relname, pg_get_viewdef(c.oid), a.attnum, a.attname, a.atttypid +FROM pg_class c JOIN pg_namespace n ON n.oid = c.relnamespace +JOIN pg_attribute a ON a.attrelid = c.oid AND a.attnum > 0 +WHERE n.nspname = 'pg_catalog' AND c.relkind = 'v' + AND (c.relname LIKE 'pg_stat_%' OR c.relname LIKE 'gp_stat_%') +ORDER BY c.relname, a.attnum +}; +my $catalog_before = $node->safe_psql('postgres', $catalog_query); +$node->safe_psql('postgres', q{ +CREATE SCHEMA maintenance; +CREATE EXTENSION vacuum_stats SCHEMA maintenance; +}); +is($node->safe_psql('postgres', $catalog_query), $catalog_before, + 'installing the extension preserves system statistics view definitions and columns'); +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM pg_proc p JOIN pg_namespace n ON n.oid = p.pronamespace +WHERE n.nspname = 'pg_catalog' + AND p.proname IN ('pg_stat_get_frozen_page_marks_cleared', + 'pg_stat_get_visible_page_marks_cleared', + 'pg_stat_get_total_vacuum_time', 'pg_stat_get_total_autovacuum_time', + 'pg_stat_get_total_analyze_time', 'pg_stat_get_total_autoanalyze_time', + 'pg_stat_get_total_vacuum_delay_time', 'pg_stat_get_total_autovacuum_delay_time', + 'pg_stat_get_vacuum_failsafe_count', + 'pg_stat_get_db_frozen_page_marks_cleared', + 'pg_stat_get_db_visible_page_marks_cleared', + 'pg_stat_get_db_total_vacuum_time', 'pg_stat_get_db_total_autovacuum_time', + 'pg_stat_get_db_total_vacuum_delay_time', 'pg_stat_get_db_total_autovacuum_delay_time', + 'pg_stat_get_db_vacuum_failsafe_count', 'pg_stat_get_db_vacuum_interrupt_count') +}), '0', 'new statistics getters are not installed as built-in functions'); + +is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'off', + 'ordinary maintenance timing is tested without extended statistics storage'); +$node->safe_psql('postgres', q{ +CREATE TABLE analyze_time (id int); +INSERT INTO analyze_time SELECT generate_series(1, 10000); +ANALYZE analyze_time; +}); +$node->poll_query_until('postgres', q{ +SELECT total_analyze_time > 0 FROM maintenance.pg_stat_vacuum_tables +WHERE relname = 'analyze_time' +}) or BAIL_OUT('ANALYZE timing report did not reach the collector'); +my $analyze_query = q{ +SELECT total_analyze_time, total_autoanalyze_time +FROM maintenance.pg_stat_vacuum_tables WHERE relname = 'analyze_time' +}; +my $analyze_before = $node->safe_psql('postgres', $analyze_query); +is($node->safe_psql('postgres', q{ +SELECT total_analyze_time > 0 AND total_autoanalyze_time = 0 +FROM maintenance.gp_stat_vacuum_tables WHERE relname = 'analyze_time' +}), 't', 'cluster view resolves extension getters in a non-default schema'); +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM maintenance.pg_stat_vacuum_database +WHERE datid = 0 AND datname IS NULL +}), '1', 'local database view exposes the shared-relation entry'); +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM maintenance.gp_stat_vacuum_database +WHERE datid = 0 AND datname IS NULL +}), '1', 'cluster database view exposes the shared entry once in utility mode'); + +$node->safe_psql('postgres', 'SELECT maintenance.vacuum_stats_reset()'); +# Drain the asynchronous reset before testing fields it must preserve. +$node->restart; +is($node->safe_psql('postgres', $analyze_query), $analyze_before, + 'dedicated vacuum reset preserves ANALYZE times'); +$node->safe_psql('postgres', 'DROP EXTENSION vacuum_stats'); +is($node->safe_psql('postgres', $catalog_query), $catalog_before, + 'dropping the extension preserves system statistics views'); +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats SCHEMA maintenance'); +is($node->safe_psql('postgres', $analyze_query), $analyze_before, + 'reinstalling the extension reads the same collector statistics'); + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/vacuum_stats--1.0.sql b/contrib/vacuum_stats/vacuum_stats--1.0.sql index 5cf0b4926c1..eea69643382 100644 --- a/contrib/vacuum_stats/vacuum_stats--1.0.sql +++ b/contrib/vacuum_stats/vacuum_stats--1.0.sql @@ -553,6 +553,36 @@ CREATE VIEW gp_stat_vacuum_database AS total_autovacuum_delay_time double precision, vacuum_interrupt_count int8); +-- +-- Resetting, for when only these counters are in the way: pg_stat_reset() +-- and pg_stat_reset_single_table_counters() throw away the rest of the +-- statistics of the database or the relation as well. +-- +-- As with the rest of the statistics, resetting acts on the node it runs on, +-- so on a cluster both the coordinator function and the segment one have to +-- be called. +-- +CREATE FUNCTION vacuum_stats_reset() RETURNS void +AS 'MODULE_PATHNAME', 'vacuum_stats_reset' +LANGUAGE C; + +CREATE FUNCTION vacuum_stats_reset(relid oid) RETURNS void +AS 'MODULE_PATHNAME', 'vacuum_stats_reset_relation' +LANGUAGE C STRICT; + +CREATE FUNCTION gp_vacuum_stats_reset() RETURNS SETOF void AS +$$ SELECT @extschema@.vacuum_stats_reset() $$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +CREATE FUNCTION gp_vacuum_stats_reset(relid oid) RETURNS SETOF void AS +$$ SELECT @extschema@.vacuum_stats_reset($1) $$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +REVOKE ALL ON FUNCTION vacuum_stats_reset() FROM PUBLIC; +REVOKE ALL ON FUNCTION vacuum_stats_reset(oid) FROM PUBLIC; +REVOKE ALL ON FUNCTION gp_vacuum_stats_reset() FROM PUBLIC; +REVOKE ALL ON FUNCTION gp_vacuum_stats_reset(oid) FROM PUBLIC; + GRANT SELECT ON pg_stat_vacuum_tables, pg_stat_vacuum_indexes, pg_stat_vacuum_database, gp_stat_vacuum_tables, gp_stat_vacuum_indexes, gp_stat_vacuum_database TO PUBLIC; diff --git a/contrib/vacuum_stats/vacuum_stats.c b/contrib/vacuum_stats/vacuum_stats.c index a84a563b612..161881a4e97 100644 --- a/contrib/vacuum_stats/vacuum_stats.c +++ b/contrib/vacuum_stats/vacuum_stats.c @@ -158,3 +158,32 @@ DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_dead_pages, dead_pages) DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_frozen, pages_frozen) DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_all_visible, pages_all_visible) DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_freeze_age_count, freeze_age_vacuum_count) + +/* + * Throw away the vacuum counters of one relation, or of the whole database, + * without touching the rest of the statistics -- which is what + * pg_stat_reset() and pg_stat_reset_single_table_counters() would do. + * + * Like the other resetting functions this acts on the node it runs on, so on + * a cluster it has to be dispatched to the segments as well; see + * gp_vacuum_stats_reset() in the extension script. + */ +PG_FUNCTION_INFO_V1(vacuum_stats_reset); +Datum +vacuum_stats_reset(PG_FUNCTION_ARGS) +{ + pgstat_reset_vacuum_stats(InvalidOid, true); + + PG_RETURN_VOID(); +} + +PG_FUNCTION_INFO_V1(vacuum_stats_reset_relation); +Datum +vacuum_stats_reset_relation(PG_FUNCTION_ARGS) +{ + Oid relid = PG_GETARG_OID(0); + + pgstat_reset_vacuum_stats(relid, false); + + PG_RETURN_VOID(); +} diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c index 7b2223a1f59..b6f3592ff9a 100644 --- a/src/backend/postmaster/pgstat.c +++ b/src/backend/postmaster/pgstat.c @@ -39,6 +39,7 @@ #include "access/twophase_rmgr.h" #include "access/xact.h" #include "access/xlog.h" +#include "catalog/catalog.h" #include "catalog/pg_database.h" #include "catalog/pg_proc.h" #include "executor/instrument.h" @@ -378,6 +379,7 @@ static void pgstat_recv_resetreplslotcounter(PgStat_MsgResetreplslotcounter *msg static void pgstat_recv_autovac(PgStat_MsgAutovacStart *msg, int len); static void pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len); static void pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len); +static void pgstat_recv_resetvacstats(PgStat_MsgResetVacstats *msg, int len); static void pgstat_recv_analyze(PgStat_MsgAnalyze *msg, int len); static void pgstat_recv_archiver(PgStat_MsgArchiver *msg, int len); static void pgstat_recv_queuestat(PgStat_MsgQueuestat *msg, int len); /* GPDB */ @@ -1690,6 +1692,36 @@ pgstat_report_vacstats(Oid tableoid, bool shared, bool isindex, pgstat_send(&msg, sizeof(msg)); } +/* ---------- + * pgstat_reset_vacuum_stats() - + * + * Tell the collector to throw away the vacuum counters of one relation of + * this database, or of all of them when resetall is true. + * ---------- + */ +void +pgstat_reset_vacuum_stats(Oid relid, bool resetall) +{ + PgStat_MsgResetVacstats msg; + + /* An invalid relation OID must never turn into a database-wide reset. */ + if (!resetall && !OidIsValid(relid)) + ereport(ERROR, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("invalid relation OID: %u", relid))); + Assert(!resetall || !OidIsValid(relid)); + + if (pgStatSock == PGINVALID_SOCKET) + return; + + pgstat_setheader(&msg.m_hdr, PGSTAT_MTYPE_RESETVACSTATS); + msg.m_databaseid = !resetall && IsSharedRelation(relid) ? + InvalidOid : MyDatabaseId; + msg.m_objectid = relid; + msg.m_resetall = resetall; + pgstat_send(&msg, sizeof(msg)); +} + /* -------- * pgstat_report_analyze() - * @@ -3785,6 +3817,10 @@ PgstatCollectorMain(int argc, char *argv[]) pgstat_recv_vacstats(&msg.msg_vacstats, len); break; + case PGSTAT_MTYPE_RESETVACSTATS: + pgstat_recv_resetvacstats(&msg.msg_resetvacstats, len); + break; + case PGSTAT_MTYPE_ANALYZE: pgstat_recv_analyze(&msg.msg_analyze, len); break; @@ -5901,6 +5937,69 @@ pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len) msg->m_stats.freeze_age_vacuum_count; } +/* ---------- + * pgstat_recv_resetvacstats() - + * + * Throw away the vacuum counters of one relation, or of the whole database + * when no relation is given. This includes the VM revision counters; + * ordinary statistics are left alone. + * ---------- + */ +static void +pgstat_recv_resetvacstats(PgStat_MsgResetVacstats *msg, int len) +{ + PgStat_StatDBEntry *dbentry; + PgStat_StatTabEntry *tabentry; + HASH_SEQ_STATUS hstat; + + dbentry = pgstat_get_db_entry(msg->m_databaseid, false); + if (!dbentry) + return; + + if (!msg->m_resetall) + { + tabentry = pgstat_get_tab_entry(dbentry, msg->m_objectid, false); + if (tabentry != NULL) + { + tabentry->frozen_page_marks_cleared = 0; + tabentry->visible_page_marks_cleared = 0; + tabentry->total_vacuum_time = 0; + tabentry->total_autovacuum_time = 0; + tabentry->total_vacuum_delay_time = 0; + tabentry->total_autovacuum_delay_time = 0; + tabentry->vacuum_failsafe_count = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&tabentry->vacuum_stats, 0, sizeof(tabentry->vacuum_stats)); + } + return; + } + + hash_seq_init(&hstat, dbentry->tables); + while ((tabentry = (PgStat_StatTabEntry *) hash_seq_search(&hstat)) != NULL) + { + tabentry->frozen_page_marks_cleared = 0; + tabentry->visible_page_marks_cleared = 0; + tabentry->total_vacuum_time = 0; + tabentry->total_autovacuum_time = 0; + tabentry->total_vacuum_delay_time = 0; + tabentry->total_autovacuum_delay_time = 0; + tabentry->vacuum_failsafe_count = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&tabentry->vacuum_stats, 0, sizeof(tabentry->vacuum_stats)); + } + dbentry->n_frozen_page_marks_cleared = 0; + dbentry->n_visible_page_marks_cleared = 0; + dbentry->total_vacuum_time = 0; + dbentry->total_autovacuum_time = 0; + dbentry->total_vacuum_delay_time = 0; + dbentry->total_autovacuum_delay_time = 0; + dbentry->vacuum_failsafe_count = 0; + dbentry->vacuum_interrupt_count = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&dbentry->n_vacuum_stats, 0, sizeof(dbentry->n_vacuum_stats)); +} + + /* ---------- * pgstat_recv_analyze() - * diff --git a/src/include/pgstat.h b/src/include/pgstat.h index 6558eeae6ce..717090a07dc 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -86,6 +86,7 @@ typedef enum StatMsgType PGSTAT_MTYPE_CONNECT, PGSTAT_MTYPE_DISCONNECT, PGSTAT_MTYPE_VACSTATS, + PGSTAT_MTYPE_RESETVACSTATS, } StatMsgType; /* ---------- @@ -734,6 +735,20 @@ typedef struct PgStat_MsgDisconnect SessionEndType m_cause; } PgStat_MsgDisconnect; +/* ---------- + * PgStat_MsgResetVacstats Sent by the backend to throw away the vacuum + * counters of one relation, or of the whole + * database when m_resetall is true. + * ---------- + */ +typedef struct PgStat_MsgResetVacstats +{ + PgStat_MsgHdr m_hdr; + Oid m_databaseid; + Oid m_objectid; + bool m_resetall; +} PgStat_MsgResetVacstats; + /* ---------- * PgStat_Msg Union over all possible messages. * ---------- @@ -754,6 +769,7 @@ typedef union PgStat_Msg PgStat_MsgAutovacStart msg_autovacuum_start; PgStat_MsgVacuum msg_vacuum; PgStat_MsgVacstats msg_vacstats; + PgStat_MsgResetVacstats msg_resetvacstats; PgStat_MsgAnalyze msg_analyze; PgStat_MsgArchiver msg_archiver; PgStat_MsgQueuestat msg_queuestat; /* GPDB */ @@ -1160,6 +1176,7 @@ extern void pgstat_report_index_vacuum_time(Relation rel, PgStat_Counter delaytime, bool is_autovacuum); extern void pgstat_report_vacstats(Oid tableoid, bool shared, bool isindex, const PgStat_VacuumStats *stats); +extern void pgstat_reset_vacuum_stats(Oid relid, bool resetall); /* count a page whose all-visible bit is being cleared */ #define pgstat_count_visible_page_marks_cleared(rel) \