diff --git a/contrib/Makefile b/contrib/Makefile
index 5ea76366363..5d8e38819d6 100644
--- a/contrib/Makefile
+++ b/contrib/Makefile
@@ -54,6 +54,7 @@ SUBDIRS = \
tsm_system_rows \
tsm_system_time \
unaccent \
+ vacuum_stats \
vacuumlo
# Cloudberry-specific additions (to ease merge pain).
diff --git a/contrib/bloom/blvacuum.c b/contrib/bloom/blvacuum.c
index 88b0a6d2900..d8873f96822 100644
--- a/contrib/bloom/blvacuum.c
+++ b/contrib/bloom/blvacuum.c
@@ -61,7 +61,7 @@ blbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
*itupPtr,
*itupEnd;
- vacuum_delay_point();
+ vacuum_delay_point(false);
buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno,
RBM_NORMAL, info->strategy);
@@ -191,7 +191,7 @@ blvacuumcleanup(IndexVacuumInfo *info, IndexBulkDeleteResult *stats)
Buffer buffer;
Page page;
- vacuum_delay_point();
+ vacuum_delay_point(false);
buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno,
RBM_NORMAL, info->strategy);
diff --git a/contrib/file_fdw/file_fdw.c b/contrib/file_fdw/file_fdw.c
index cce94a5b335..5e870115661 100644
--- a/contrib/file_fdw/file_fdw.c
+++ b/contrib/file_fdw/file_fdw.c
@@ -1170,7 +1170,7 @@ file_acquire_sample_rows(Relation onerel, int elevel,
for (;;)
{
/* Check for user-requested abort or sleep */
- vacuum_delay_point();
+ vacuum_delay_point(true);
/* Fetch next row */
MemoryContextReset(tupcontext);
diff --git a/contrib/vacuum_stats/.gitignore b/contrib/vacuum_stats/.gitignore
new file mode 100644
index 00000000000..5dcb3ff9723
--- /dev/null
+++ b/contrib/vacuum_stats/.gitignore
@@ -0,0 +1,4 @@
+# Generated subdirectories
+/log/
+/results/
+/tmp_check/
diff --git a/contrib/vacuum_stats/Makefile b/contrib/vacuum_stats/Makefile
new file mode 100644
index 00000000000..aee5f608665
--- /dev/null
+++ b/contrib/vacuum_stats/Makefile
@@ -0,0 +1,29 @@
+# contrib/vacuum_stats/Makefile
+
+MODULE_big = vacuum_stats
+OBJS = vacuum_stats.o
+
+EXTENSION = vacuum_stats
+DATA = vacuum_stats--1.0.sql
+PGFILEDESC = "vacuum_stats - per-relation and per-database vacuum statistics"
+
+TAP_TESTS = 1
+
+ifdef USE_PGXS
+PG_CONFIG = pg_config
+PGXS := $(shell $(PG_CONFIG) --pgxs)
+include $(PGXS)
+else
+subdir = contrib/vacuum_stats
+top_builddir = ../..
+include $(top_builddir)/src/Makefile.global
+include $(top_srcdir)/contrib/contrib-global.mk
+endif
+
+check-tap:
+ $(prove_check)
+
+installcheck-tap:
+ $(prove_installcheck)
+
+.PHONY: check-tap installcheck-tap
diff --git a/contrib/vacuum_stats/README.md b/contrib/vacuum_stats/README.md
new file mode 100644
index 00000000000..4f1252dc2ec
--- /dev/null
+++ b/contrib/vacuum_stats/README.md
@@ -0,0 +1,139 @@
+# vacuum_stats
+
+`vacuum_stats` describes the work done by VACUUM for tables, indexes and
+whole databases: how many tuples it removes, what remains to be cleaned,
+how it changes page visibility, and how much time it spends. These counters
+help evaluate the results and cost of vacuuming over time, alongside the
+existing vacuum counts and timestamps.
+
+Collection of extended work counters is enabled with `track_vacuum_statistics = on` in the server
+configuration and requires a restart. Set it consistently on the coordinator
+and segments. When enabled, the extended vacuum counters and database totals
+are stored with ordinary relation and database statistics. When disabled, their storage
+is omitted and those extended fields return zero. Restarting with tracking
+disabled discards the extended counters while preserving ordinary statistics,
+including vacuum times, VM clearings, failsafe and interruption counts.
+
+The `visible_page_marks_cleared` and `frozen_page_marks_cleared` counters describe
+visibility-map changes caused by data modifications. They follow `track_counts`
+and are collected and retained independently of `track_vacuum_statistics`.
+The dedicated vacuum-statistics reset functions also reset these counters,
+vacuum times and failsafe counts. Database-wide vacuum reset also
+clears `pg_stat_vacuum_database.vacuum_interrupt_count`; relation reset preserves
+this database total. The reset functions preserve ordinary access counters,
+maintenance counts and timestamps, and ANALYZE times.
+
+Install with `CREATE EXTENSION vacuum_stats`. All new SQL functions and
+views belong to the extension's schema. Existing system views and built-in
+function OIDs remain unchanged. VM clearings, maintenance times, failsafe and
+interruption counts follow `track_counts` independently of
+`track_vacuum_statistics`; the extension reads their ordinary collector entries.
+
+Cost-based delay timing additionally requires `track_cost_delay_timing = on`.
+It is disabled by default and can be changed for a session without restarting.
+ANALYZE elapsed times are recorded separately from VACUUM times. ANALYZE
+sampling delays do not contribute to the VACUUM delay counters.
+
+The counters accumulate until reset, except `total_file_segs`, which records
+the state observed by the last completed AO VACUUM. To examine a particular period, compare
+two readings without an intervening reset or change in tracking configuration.
+
+## What the counters measure
+
+| Counter | Meaning |
+|---------|---------|
+| `tuples_deleted` | Table tuples or index entries removed by VACUUM. |
+| `dead_tuples` | Heap tuples found dead but not yet removable; for AO tables, hidden tuples remaining after vacuum. |
+| `pages_deleted` | Pages truncated from a heap, newly deleted index pages, or space freed from AO segment files expressed in blocks. |
+| `bytes_removed` | Bytes physically truncated from heap or AO table files, including AO tails left by aborted inserts. Index page reuse does not increase this counter. |
+| `dead_pages` | Heap pages containing unremovable dead tuples; for indexes, deleted pages not yet available for reuse. |
+| `pages_frozen` | Heap pages on which VACUUM froze at least one tuple. |
+| `pages_all_visible` | Heap pages VACUUM marked all-visible. |
+| `visible_page_marks_cleared` | Clearings of the all-visible flag, usually caused by data changes. |
+| `frozen_page_marks_cleared` | Clearings of the all-frozen flag, usually caused by data changes. |
+| `freeze_age_vacuum_count` | Heap vacuum runs made aggressive by transaction or multixact freeze age. This includes `VACUUM FREEZE`; forcing page scanning alone does not increment it. |
+| `vacuum_failsafe_count` | Completed heap vacuum runs that entered failsafe mode to avoid transaction or multixact wraparound. Aggressive scanning alone does not increment it. |
+| `total_vacuum_time`, `total_autovacuum_time` | Elapsed time in milliseconds, separately for manual VACUUM and autovacuum. |
+| `total_vacuum_delay_time`, `total_autovacuum_delay_time` | Cost-based delays in milliseconds, included in the corresponding elapsed time. |
+| `total_analyze_time`, `total_autoanalyze_time` | Table ANALYZE time in milliseconds, separately for manual and automatic runs. |
+| `vacuum_interrupt_count` | Database count of heap vacuums interrupted by an ERROR, including cancellation, while the vacuum error callback is installed. |
+
+`total_file_segs` in the table views records the number of AO segment metadata
+entries observed after the last VACUUM, including empty and awaiting-drop
+segments. For AO column tables it counts logical segments, not each column
+file. A subsequent VACUUM replaces this value instead of adding to it. It is
+zero before the first report, after reset, and for heap tables. Use it with
+`bytes_removed` and remaining hidden tuples to understand compaction results.
+
+The other fields are cumulative counters, not a snapshot of the table's current
+contents. In particular, successive runs can count the same unremovable
+tuple or page again. Visibility flags can also be set and cleared repeatedly.
+A counter difference measures work or observations during the interval,
+not necessarily a number of distinct tuples or pages.
+
+## How to use them
+
+- **Find expensive relations.** Compare increases in `total_vacuum_time` and
+ `total_autovacuum_time` between tables and indexes over the same interval. Relate that time to tuples
+ removed and pages reclaimed to see where maintenance time is spent.
+ Low tuple removal alone does not imply wasted work: vacuum also freezes
+ tuples and maintains visibility information.
+- **Find work that cannot finish.** Repeated increases in heap `dead_tuples`
+ and `dead_pages` show that vacuum keeps encountering data it cannot remove.
+ Check for old snapshots or long-running transactions before increasing
+ vacuum frequency.
+- **Separate throttling from other costs.** Compare `total_vacuum_delay_time` with
+ `total_vacuum_time`, and the corresponding autovacuum counters. A large share
+ spent in cost-based delays helps explain a long run. The remaining time includes execution and other waits; it is
+ not a measurement of CPU time.
+- **Understand visibility and freezing work.** Compare `pages_all_visible`
+ with `visible_page_marks_cleared` to see how often data changes undo visibility
+ work. Use `pages_frozen`, `frozen_page_marks_cleared` and
+ `freeze_age_vacuum_count` to understand freezing activity and aggressive
+ scans. Zero frozen pages can be normal when no tuples need freezing.
+- **Compare segments.** Differences in work and time for the same relation
+ can help identify uneven data distribution or different execution costs.
+ Compare both quantities: a slower segment is not necessarily processing
+ more data. Summed segment time represents accumulated work, not the
+ wall-clock duration of a distributed VACUUM.
+
+## Tables, indexes and AO
+
+A table's vacuum time includes its index maintenance. Database totals
+include table work without adding the index counters again. Use the index
+figures to understand that part of the cost, rather than adding them to
+table totals. Deleted index pages can become reusable within the index;
+they do not necessarily represent space returned to the filesystem.
+
+For AO row and column tables, `tuples_deleted` measures rows discarded by
+compaction, and `dead_tuples` counts hidden rows left afterwards. Hidden
+rows can remain when compaction is disabled or a segment file is not
+eligible for compaction. AO index cleanup can remove entries for relocated
+live rows as well as deleted rows, so its tuple count can exceed the table's.
+AO `pages_deleted` expresses truncated bytes in blocks, rounded up;
+`bytes_removed` preserves the exact byte count. Freed space can include live
+rows moved to new segment files, so this is not the net reduction in table size.
+AO elapsed time covers the interval from the first phase seen by the
+reporting worker to final cleanup, including gaps between those phases.
+
+Heap page visibility, freezing and failsafe counters do not apply to the AO table
+itself. Its auxiliary heap relations have their own statistics. Indexes
+have tuple-removal, page-deletion and timing counters, but no heap visibility
+or freezing work.
+
+The `pg_stat_vacuum_tables`, `pg_stat_vacuum_indexes` and
+`pg_stat_vacuum_database` views expose local statistics. Their
+`gp_stat_vacuum_*` counterparts include the coordinator and segments,
+identified by `gp_segment_id`. In utility mode there is no dispatch: these
+views return the connected node's local statistics once, with its segment ID.
+
+The database views include `datid = 0`, with a null `datname`, for shared
+relations. `vacuum_interrupt_count` does not include VACUUM FULL, failures
+before the heap callback is installed, or AO parent compaction. Auxiliary
+heap vacuums are counted independently. An interrupted run does not report
+its usual completion counters, so check errors when successful-run totals
+alone do not explain maintenance activity.
+
+The on-disk statistics format changes, so older saved statistics are discarded
+on first start. The system catalog version is unchanged; these statistics do
+not require a new cluster or replacement of system views.
diff --git a/contrib/vacuum_stats/t/001_vacuum_statistics.pl b/contrib/vacuum_stats/t/001_vacuum_statistics.pl
new file mode 100644
index 00000000000..8fc4d92558c
--- /dev/null
+++ b/contrib/vacuum_stats/t/001_vacuum_statistics.pl
@@ -0,0 +1,822 @@
+# Licensed to the Apache Software Foundation (ASF) under one
+# or more contributor license agreements. See the NOTICE file
+# distributed with this work for additional information
+# regarding copyright ownership. The ASF licenses this file
+# to you under the Apache License, Version 2.0 (the
+# "License"); you may not use this file except in compliance
+# with the License. You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing,
+# software distributed under the License is distributed on an
+# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+# KIND, either express or implied. See the License for the
+# specific language governing permissions and limitations
+# under the License.
+
+# Test vacuum counter semantics, controls and lifecycle in one TAP suite.
+# Adapt the repeatable-read and visibility-map scenarios from v44:
+# https://www.postgresql.org/message-id/flat/cb305107-5935-4c34-9847-6ff0fef89f06%40yandex.ru#8b78d4ab399e37afc5eb2cee8a54d2f6
+# This branch has a UDP collector and the older pg_stat_vacuum_* API:
+# dead_tuples corresponds to recently_dead_tuples, and the page counters
+# expose freezing and VM transitions. Polling uses a fresh connection, so
+# it does not reuse a cached statistics snapshot. Core VM clearing semantics
+# are covered by 004_visibility_map_stats.pl; here check vacuum
+# reports, the extension views, controls and lifecycle.
+# PostgresNode runs in utility/maintenance mode; use the local views.
+# Read ordinary counters from pg_stat_all_tables_internal: the public
+# pg_stat_all_tables gathers segment data and has no user-table rows here.
+
+use strict;
+use warnings;
+use PostgresNode;
+use TestLib;
+use Test::More;
+
+my $node = get_new_node('vacuum_stats');
+$node->init;
+$node->append_conf('postgresql.conf', q{
+autovacuum = off
+track_counts = on
+vacuum_cost_delay = 0
+});
+$node->start;
+$node->safe_psql('postgres', q{
+CREATE EXTENSION vacuum_stats;
+CREATE TABLE vstat_barrier (id int) WITH (autovacuum_enabled = off);
+});
+
+my @counter_names = qw(tuples_deleted dead_tuples pages_deleted bytes_removed dead_pages
+ pages_frozen pages_all_visible frozen_page_marks_cleared visible_page_marks_cleared
+ freeze_age_vacuum_count vacuum_failsafe_count total_vacuum_time total_autovacuum_time
+ total_vacuum_delay_time total_autovacuum_delay_time);
+my $all_counters = join(', ', @counter_names);
+my $all_zero = join(' AND ', map { "$_ = 0" } @counter_names);
+my $vacuum_zero = join(' AND ', map { "$_ = 0" } grep { !/_page_marks_cleared$|^total_.*time$|^vacuum_failsafe_count$/ } @counter_names);
+my $vm_columns = 'frozen_page_marks_cleared, visible_page_marks_cleared';
+
+sub wait_for_stats
+{
+ my ($sql, $description, $expected) = @_;
+ $expected = 't' unless defined $expected;
+ $node->poll_query_until('postgres', $sql, $expected)
+ or BAIL_OUT("timed out waiting for $description: $sql");
+}
+
+# VACUUM sends vacuum_count before its extended counters. Instead of
+# using that earlier report, send ANALYZE from the same backend AFTER the
+# command and wait for its report. ANALYZE adds no vacuum counters, so it
+# also works for database totals, disabled tracking, resets and VACUUM FULL.
+# DML counters are buffered separately; wait for the affected table's DML
+# report explicitly when checking VM clearing counters.
+sub run_and_wait
+{
+ my ($sql) = @_;
+ my $count = $node->safe_psql('postgres',
+ "SELECT analyze_count FROM pg_stat_all_tables_internal WHERE relname = 'vstat_barrier'");
+ BAIL_OUT("missing local ANALYZE count for vstat_barrier")
+ unless $count =~ /^\d+$/;
+ my ($result, $stdout, $stderr) = $node->psql('postgres',
+ "$sql;\nANALYZE vstat_barrier;", on_error_die => 1);
+ wait_for_stats(
+ "SELECT analyze_count = $count + 1 FROM pg_stat_all_tables_internal WHERE relname = 'vstat_barrier'",
+ "collector report after $sql");
+ return $stderr;
+}
+
+sub wait_for_updates
+{
+ my ($table, $count) = @_;
+ wait_for_stats(
+ "SELECT n_tup_upd = $count FROM pg_stat_all_tables_internal WHERE relname = '$table'",
+ "UPDATE report for $table");
+}
+
+sub populated_pages
+{
+ my ($table, $predicate) = @_;
+ $predicate ||= 'true';
+ return $node->safe_psql('postgres',
+ "SELECT count(DISTINCT split_part(ctid::text, ',', 1)) FROM $table WHERE $predicate");
+}
+
+sub vacuum_table
+{
+ my ($table, $options, $settings) = @_;
+ $options ||= '';
+ $settings ||= '';
+ return run_and_wait("$settings VACUUM $options $table");
+}
+
+sub counters
+{
+ my ($table, $columns) = @_;
+ $columns ||= $all_counters;
+ return $node->safe_psql('postgres',
+ "SELECT $columns FROM pg_stat_vacuum_tables WHERE relname = '$table'");
+}
+
+sub index_counters
+{
+ my ($index, $columns) = @_;
+ $columns ||= $all_counters;
+ return $node->safe_psql('postgres',
+ "SELECT $columns FROM pg_stat_vacuum_indexes WHERE indexrelname = '$index'");
+}
+
+sub database_counters
+{
+ my ($columns) = @_;
+ $columns ||= $all_counters;
+ return $node->safe_psql('postgres',
+ "SELECT $columns FROM pg_stat_vacuum_database WHERE datname = 'postgres'");
+}
+
+# Read the same statistics snapshot before measuring its hash allocations.
+sub snapshot_bytes
+{
+ return $node->safe_psql('postgres', q{
+BEGIN;
+DO $$ BEGIN PERFORM count(*) FROM pg_stat_all_tables_internal WHERE n_tup_ins >= 0; END $$;
+SELECT sum(used_bytes) FROM pg_backend_memory_contexts
+WHERE name IN ('Databases hash', 'Per-database table');
+COMMIT;
+});
+}
+
+subtest 'tracking disabled and enabled' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_off (id int) WITH (autovacuum_enabled = off);
+INSERT INTO vstat_off SELECT generate_series(1, 1000);
+DELETE FROM vstat_off;
+});
+ my $verbose = vacuum_table('vstat_off', 'VERBOSE');
+ like($verbose, qr/table "vstat_off": vacuum statistics.*?cost-based delay: 0\.000 ms/s,
+ 'VERBOSE reports measurements even with statistics tracking disabled');
+ is(counters('vstat_off', $vacuum_zero), 't',
+ 'extended counters stay zero while tracking is disabled');
+ is(counters('vstat_off', 'total_vacuum_time > 0, total_autovacuum_time = 0'),
+ 't|t', 'core timing remains enabled independently of extended counters');
+ is($node->safe_psql('postgres',
+ "SELECT vacuum_count FROM pg_stat_all_tables_internal WHERE relname = 'vstat_off'"),
+ '1', 'ordinary vacuum statistics are still collected');
+ for my $am ('ao_row', 'ao_column')
+ {
+ my $table = "vstat_off_$am";
+ $node->safe_psql('postgres', qq{
+CREATE TABLE $table (id int) USING $am;
+CREATE INDEX ${table}_idx ON $table (id);
+INSERT INTO $table SELECT generate_series(1, 10000);
+DELETE FROM $table WHERE id % 2 = 0;
+});
+ my $verbose = vacuum_table($table, 'VERBOSE', q{
+SET track_cost_delay_timing = on;
+SET vacuum_cost_delay = '1ms';
+SET vacuum_cost_limit = 1;
+});
+ is(counters($table, "$vacuum_zero AND total_file_segs = 0"), 't',
+ "$am extended counters and segment snapshot stay zero with tracking off");
+ is(index_counters("${table}_idx", $vacuum_zero), 't',
+ "$am index work counters stay zero with tracking off");
+ is(counters($table, 'total_vacuum_time > 0 AND total_vacuum_delay_time > 0'),
+ 't', "$am table timing remains independent of extended tracking");
+ is(index_counters("${table}_idx", 'total_vacuum_time > 0 AND total_vacuum_delay_time > 0'),
+ 't', "$am compaction index timing remains independent of extended tracking");
+ like($verbose, qr/[1-9]\d* bytes truncated; \d+ file segments remain\./,
+ "$am VERBOSE retains byte and segment measurements with tracking off");
+
+ # The following run has no obsolete segments and uses scan_index().
+ my $index_time = index_counters("${table}_idx", 'total_vacuum_time');
+ vacuum_table($table);
+ is(index_counters("${table}_idx", "total_vacuum_time > $index_time"), 't',
+ "$am cleanup-only index timing advances with extended tracking off");
+ is(index_counters("${table}_idx", $vacuum_zero), 't',
+ "$am cleanup-only index work counters remain disabled");
+ }
+ is(database_counters($vacuum_zero), 't',
+ 'heap and AO work do not populate extended database totals with tracking off');
+ # VM changes are ordinary DML statistics even with vacuum tracking off.
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_vm_off (id int PRIMARY KEY) WITH (autovacuum_enabled = off);
+INSERT INTO vstat_vm_off SELECT generate_series(1, 1000);
+});
+ vacuum_table('vstat_vm_off', 'FREEZE');
+ my $pages = $node->safe_psql('postgres',
+ "SELECT pg_relation_size('vstat_vm_off') / current_setting('block_size')::bigint");
+ my ($db_frozen, $db_visible) = split /\|/, database_counters($vm_columns);
+ $node->safe_psql('postgres', 'UPDATE vstat_vm_off SET id = id + 1000');
+ wait_for_updates('vstat_vm_off', 1000);
+ is(counters('vstat_vm_off', $vm_columns), "$pages|$pages",
+ 'DML counts exact VM clearings while vacuum tracking is off');
+ is(database_counters("frozen_page_marks_cleared - $db_frozen, visible_page_marks_cleared - $db_visible"),
+ "$pages|$pages", 'database VM totals receive the same disabled-tracking DML report');
+ is(counters('vstat_vm_off', $vacuum_zero), 't',
+ 'VM collection does not enable vacuum counters');
+ $node->restart;
+ is(counters('vstat_vm_off', $vm_columns), "$pages|$pages",
+ 'VM counters survive a clean restart with tracking off');
+ # dynahash allocates entries in batches. With only a few relations,
+ # a larger entry can use a slightly smaller batch and appear cheaper.
+ # Populate enough ordinary entries to exceed that allocation rounding;
+ # none of these relations has been vacuumed.
+ $node->safe_psql('postgres', q{
+CREATE SCHEMA vstat_memory;
+DO $$ BEGIN
+ FOR i IN 1..1024 LOOP
+ EXECUTE format('CREATE TABLE vstat_memory.t%s (id int) WITH (autovacuum_enabled = off)', i);
+ EXECUTE format('INSERT INTO vstat_memory.t%s VALUES (1)', i);
+ END LOOP;
+END $$;
+});
+ wait_for_stats(q{
+SELECT count(*) = 1024 AND bool_and(n_tup_ins = 1)
+FROM pg_stat_all_tables_internal WHERE schemaname = 'vstat_memory'
+}, 'ordinary statistics for the memory-allocation test');
+ my $off_bytes = snapshot_bytes();
+ is($node->safe_psql('postgres',
+ "SELECT context FROM pg_settings WHERE name = 'track_vacuum_statistics'"),
+ 'postmaster', 'tracking is fixed at server startup');
+ my ($result, $stdout, $stderr) = $node->psql('postgres',
+ 'SET track_vacuum_statistics = on');
+ is($result, 3, 'a session cannot enable tracking');
+ like($stderr, qr/cannot be changed without restarting the server/,
+ 'the error explains that a restart is required');
+ $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = on');
+ $node->reload;
+ wait_for_stats(q{SELECT pending_restart FROM pg_settings WHERE name = 'track_vacuum_statistics'},
+ 'configuration reload to notice the startup setting');
+ is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'off',
+ 'reload leaves tracking disabled');
+ $node->restart;
+ is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'on',
+ 'restart enables tracking');
+ is(counters('vstat_vm_off', $vm_columns), "$pages|$pages",
+ 'enabling vacuum tracking preserves VM counters collected while off');
+ is(counters('vstat_off', $vacuum_zero), 't',
+ 'new vacuum blocks start at zero when reading ordinary-only statistics');
+ cmp_ok(snapshot_bytes(), '>', $off_bytes,
+ 'enabled snapshots allocate vacuum counters with ordinary relation and DB entries');
+
+};
+
+subtest 'removed tuples and truncated pages' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_heap (id int PRIMARY KEY) WITH (autovacuum_enabled = off);
+INSERT INTO vstat_heap SELECT generate_series(1, 10000);
+DELETE FROM vstat_heap WHERE id % 2 = 0;
+});
+ is(counters('vstat_heap', $all_zero), 't',
+ 'all counters are zero before the first vacuum');
+ # Heap insertion can pre-extend the relation with empty tail pages.
+ # Disable truncation in this run to make its zero counter deterministic.
+ my $verbose = vacuum_table('vstat_heap', '(VERBOSE, TRUNCATE false)');
+ is(counters('vstat_heap', 'tuples_deleted, dead_tuples, pages_deleted, dead_pages'),
+ '5000|0|0|0', 'vacuum removes exactly half the rows without truncation');
+ like($verbose, qr/pages with dead tuples not yet removable: 0\n/,
+ 'VERBOSE reports zero pages retaining dead tuples');
+ like($verbose, qr/pages with tuples frozen: 0\n/,
+ 'VERBOSE reports zero pages frozen with the default freeze age');
+ my $visible = counters('vstat_heap', 'pages_all_visible');
+ like($verbose, qr/pages marked all-visible: \Q$visible\E\n/,
+ 'VERBOSE all-visible count matches the first vacuum report');
+ like($verbose, qr/scanned index "vstat_heap_pkey".*?cost-based delay: 0\.000 ms/s,
+ 'VERBOSE reports the index cost delay');
+ is(index_counters('vstat_heap_pkey', 'tuples_deleted, pages_deleted'),
+ '5000|0', 'every other index key remains; no index pages are deleted');
+ is(counters('vstat_heap', 'total_vacuum_time > 0, total_autovacuum_time = 0, total_vacuum_delay_time = 0'),
+ 't|t|t', 'heap vacuum takes time but has no cost delay');
+ is(index_counters('vstat_heap_pkey', 'total_vacuum_time > 0, total_autovacuum_time = 0, total_vacuum_delay_time = 0'),
+ 't|t|t', 'index vacuum takes time but has no cost delay');
+
+ $node->safe_psql('postgres', 'DELETE FROM vstat_heap');
+ my $pages_before = $node->safe_psql('postgres',
+ "SELECT pg_relation_size('vstat_heap') / current_setting('block_size')::bigint");
+ my $quiet = vacuum_table('vstat_heap');
+ unlike($quiet, qr/vacuum statistics|pages marked all-visible|cost-based delay/,
+ 'ordinary VACUUM does not emit VERBOSE statistics at the default message level');
+ is(counters('vstat_heap', 'tuples_deleted, dead_tuples, pages_deleted'),
+ "10000|0|$pages_before", 'the second vacuum removes the remaining rows and truncates every heap page');
+ is(counters('vstat_heap', "bytes_removed = pages_deleted * current_setting('block_size')::bigint"),
+ 't', 'heap truncation reports exact bytes');
+ is(index_counters('vstat_heap_pkey', 'tuples_deleted, pages_deleted > 0'),
+ '10000|t', 'index tuple and page deletion counters accumulate');
+};
+
+subtest 'cluster views return local rows once in utility mode' => sub {
+ is($node->safe_psql('postgres', 'SHOW gp_role'), 'utility',
+ 'this scenario uses a direct utility connection');
+ for my $view (
+ ['tables', 'relid, schemaname, relname', "relname = 'vstat_heap'"],
+ ['indexes', 'relid, indexrelid, schemaname, relname, indexrelname',
+ "relname = 'vstat_heap'"],
+ ['database', 'datid, datname', "datname = 'postgres'"])
+ {
+ my ($suffix, $keys, $filter) = @$view;
+ my $columns = "$keys, $all_counters";
+ $columns .= ', total_analyze_time, total_autoanalyze_time' if $suffix eq 'tables';
+ $columns .= ', total_file_segs' if $suffix eq 'tables';
+ $columns .= ', vacuum_interrupt_count' if $suffix eq 'database';
+ my $local = "SELECT $columns FROM pg_stat_vacuum_$suffix WHERE $filter";
+ my $cluster = "SELECT $columns FROM gp_stat_vacuum_$suffix WHERE $filter";
+ is($node->safe_psql('postgres', qq{
+SELECT count(*) = 1 AND bool_and(gp_segment_id = gp_execution_segment())
+FROM gp_stat_vacuum_$suffix WHERE $filter
+}), 't', "$suffix view returns one row identified by the connected node");
+ is($node->safe_psql('postgres', qq{
+SELECT NOT EXISTS (
+ ($cluster EXCEPT ALL $local)
+ UNION ALL
+ ($local EXCEPT ALL $cluster)
+)
+}), 't', "$suffix cluster and local views have identical rows and counters");
+ }
+};
+
+subtest 'a repeatable-read snapshot prevents removal' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_snapshot (id int PRIMARY KEY, val int)
+ WITH (autovacuum_enabled = off);
+INSERT INTO vstat_snapshot SELECT i, i FROM generate_series(1, 1000) g(i);
+});
+ my $dead_pages = populated_pages('vstat_snapshot', 'id > 900');
+ my ($in, $out) = ('', '');
+ my $timer = IPC::Run::timeout($TestLib::timeout_default);
+ my $reader = $node->background_psql('postgres', \$in, \$out, $timer);
+ $out = '';
+ $in = "BEGIN ISOLATION LEVEL REPEATABLE READ;\n"
+ . "SELECT count(*) FROM vstat_snapshot;\n\\echo snapshot_ready\n";
+ pump_until($reader, $timer, \$out, qr/^snapshot_ready\r?$/m)
+ or BAIL_OUT('reader did not acquire its snapshot');
+ like($out, qr/^1000\r?$/m, 'reader sees all original rows');
+
+ # Updating the indexed column prevents HOT, making index removals exact.
+ $node->safe_psql('postgres', 'UPDATE vstat_snapshot SET id = id + 1000 WHERE id > 900');
+ my $verbose = vacuum_table('vstat_snapshot', 'VERBOSE');
+ is(counters('vstat_snapshot', 'tuples_deleted, dead_tuples, dead_pages, pages_frozen'),
+ "0|100|$dead_pages|0", '100 old tuple versions remain on exactly the affected pages');
+ like($verbose, qr/pages with dead tuples not yet removable: \Q$dead_pages\E\n/,
+ 'VERBOSE reports the pages held back by the snapshot');
+ is(index_counters('vstat_snapshot_pkey', 'tuples_deleted'),
+ '0', 'index entries needed by the reader are retained');
+
+ $in = "COMMIT;\n\\q\n";
+ $reader->finish;
+ vacuum_table('vstat_snapshot');
+ is(counters('vstat_snapshot', 'tuples_deleted, dead_tuples, pages_frozen'),
+ '100|100|0', 'after commit, 100 versions are removed; cumulative dead_tuples stays at 100');
+ is(index_counters('vstat_snapshot_pkey', 'tuples_deleted'), '100',
+ 'after commit, the index removes exactly the 100 obsolete entries');
+};
+
+subtest 'heap page reports and core VM counters in extension views' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_vm (id int PRIMARY KEY, val int)
+ WITH (autovacuum_enabled = off, fillfactor = 50);
+INSERT INTO vstat_vm SELECT i, i FROM generate_series(1, 5000) g(i);
+});
+ my $pages = populated_pages('vstat_vm');
+ my $verbose = vacuum_table('vstat_vm', 'VERBOSE',
+ 'SET vacuum_freeze_min_age = 1000000000; SET vacuum_freeze_table_age = 1000000000;');
+ is(counters('vstat_vm', 'pages_frozen, pages_all_visible, frozen_page_marks_cleared, visible_page_marks_cleared'),
+ "0|$pages|0|0", 'ordinary vacuum marks each populated page visible without freezing it');
+ like($verbose, qr/pages with tuples frozen: 0\n/,
+ 'VERBOSE reports no freezing with a high freeze age');
+ like($verbose, qr/pages marked all-visible: \Q$pages\E\n/,
+ 'VERBOSE reports the exact number of newly visible pages');
+
+ # Leave enough room for new versions on the same pages and prevent HOT.
+ # The distinct counter values detect swapped columns in extension views.
+ $node->safe_psql('postgres', 'UPDATE vstat_vm SET id = id + 10000');
+ wait_for_updates('vstat_vm', 5000);
+ # Do not scan the heap here: that could prune the obsolete versions before
+ # VACUUM gets to count their removal.
+ is($node->safe_psql('postgres',
+ "SELECT pg_relation_size('vstat_vm') / current_setting('block_size')::bigint"),
+ $pages, 'updated versions fit on the original pages');
+ is(counters('vstat_vm', 'frozen_page_marks_cleared, visible_page_marks_cleared'),
+ "0|$pages", 'extension view reports only all-visible clearings for unfrozen pages');
+
+ $verbose = vacuum_table('vstat_vm', '(FREEZE, VERBOSE)');
+ my $visible = 2 * $pages;
+ is(counters('vstat_vm', 'tuples_deleted, pages_frozen, pages_all_visible'),
+ "5000|$pages|$visible", 'FREEZE reports removals, freezing and restored visibility separately');
+ is(counters('vstat_vm', 'frozen_page_marks_cleared, visible_page_marks_cleared'),
+ "0|$pages", 'restoring VM flags adds no clearings');
+ like($verbose, qr/pages with tuples frozen: \Q$pages\E\n/,
+ 'VERBOSE reports exactly one freeze per populated page');
+ like($verbose, qr/pages marked all-visible: \Q$pages\E\n/,
+ 'VERBOSE reports this run\'s visibility work, not the cumulative total');
+ is(counters('vstat_vm', 'freeze_age_vacuum_count'), '1',
+ 'FREEZE counts the vacuum made aggressive by the freeze age');
+ like($verbose, qr/aggressive scan required by freeze age: yes/,
+ 'VERBOSE explains freeze-age-driven aggressive scanning');
+
+ $verbose = vacuum_table('vstat_vm', '(FREEZE, VERBOSE)');
+ is(counters('vstat_vm', 'pages_frozen, pages_all_visible'),
+ "$pages|$visible", 'another FREEZE adds no freezing or visibility work');
+ like($verbose, qr/pages with tuples frozen: 0\n/,
+ 'VERBOSE reports no repeated freezing');
+ like($verbose, qr/pages marked all-visible: 0\n/,
+ 'VERBOSE reports no repeated visibility changes');
+
+ $node->safe_psql('postgres', 'DELETE FROM vstat_vm');
+ wait_for_stats(q{
+SELECT n_tup_del = 5000 FROM pg_stat_all_tables_internal WHERE relname = 'vstat_vm'
+}, 'DELETE report for frozen pages');
+ is(counters('vstat_vm', 'frozen_page_marks_cleared, visible_page_marks_cleared'),
+ "$pages|$visible", 'extension view exposes both core VM counters with distinct totals');
+};
+
+subtest 'append-optimized compaction' => sub {
+ for my $am ('ao_row', 'ao_column')
+ {
+ my $table = "vstat_$am";
+ $node->safe_psql('postgres', qq{
+CREATE TABLE $table (id int, val int) USING $am;
+CREATE INDEX ${table}_idx ON $table (id);
+INSERT INTO $table SELECT i, i FROM generate_series(1, 10000) g(i);
+DELETE FROM $table WHERE id % 2 = 0;
+});
+ # Preserve the zero-freeze-age case: AO still must not count a
+ # freeze-age vacuum, unlike its auxiliary heap relations.
+ my $verbose = vacuum_table($table, 'VERBOSE', 'SET vacuum_freeze_table_age = 0;');
+ is(counters($table, 'tuples_deleted, dead_tuples, pages_frozen, pages_all_visible, freeze_age_vacuum_count'),
+ '5000|0|0|0|0', "$am compaction removes 5000 rows and has no heap VM or freezing work");
+ # Compaction relocates surviving tuples, so all original index
+ # entries, including those of surviving rows, become obsolete.
+ is(index_counters("${table}_idx", 'tuples_deleted'),
+ '10000', "$am index cleanup removes the old TIDs");
+ is(counters($table, 'total_vacuum_time > 0, total_autovacuum_time = 0, total_vacuum_delay_time = 0'),
+ 't|t|t', "$am compaction takes time without cost delay");
+ my $bytes = counters($table, 'bytes_removed');
+ my $segrel = $node->safe_psql('postgres',
+ "SELECT segrelid::regclass FROM pg_appendonly WHERE relid = '$table'::regclass");
+ my $segs = $node->safe_psql('postgres', "SELECT count(*) FROM $segrel");
+ is(counters($table, 'total_file_segs'), $segs,
+ "$am reports the remaining segment metadata entries");
+ is($node->safe_psql('postgres',
+ "SELECT total_file_segs FROM gp_stat_vacuum_tables WHERE relname = '$table'"),
+ $segs, "$am cluster view exposes the segment snapshot in utility mode");
+ like($verbose, qr/\Q$bytes bytes truncated; $segs file segments remain.\E/,
+ "$am VERBOSE byte and segment counts match the report");
+ like($verbose,
+ qr/append-optimized table "\Q$table\E": vacuum statistics\nDETAIL: \Q0 dead tuples remain.\E\n\d+ bytes truncated; \d+ file segments remain\.\nelapsed: \d+\.\d{3} ms, cost-based delay: 0\.000 ms/,
+ "$am VERBOSE reports remaining dead tuples and accumulated phase time");
+ is(counters($table, 'pages_deleted > 0, dead_pages, frozen_page_marks_cleared, visible_page_marks_cleared'),
+ 't|0|0|0', "$am reports freed space without heap page or VM counters");
+
+ # Without obsolete segment files, AO still runs index cleanup.
+ # Its time must be reported even if the AM returns no page statistics.
+ my $index_time = index_counters("${table}_idx", 'total_vacuum_time');
+ vacuum_table($table);
+ is(counters($table, 'bytes_removed, total_file_segs'), "$bytes|$segs",
+ "$am idle vacuum neither adds reclaimed bytes nor sums segment counts");
+ is(index_counters("${table}_idx", "tuples_deleted, total_vacuum_time > $index_time, total_vacuum_delay_time = 0"),
+ '10000|t|t', "$am reports index cleanup without deleting more TIDs");
+
+ $node->safe_psql('postgres', "DELETE FROM $table WHERE id <= 200");
+ vacuum_table($table, '', 'SET gp_appendonly_compaction = off;');
+ is(counters($table, 'tuples_deleted, dead_tuples'),
+ '5000|100', "$am counts hidden rows left when compaction is disabled");
+ }
+};
+
+subtest 'AO truncate reports exact bytes beyond the 32-bit boundary' => sub {
+ my $block_size = $node->safe_psql('postgres', 'SHOW block_size');
+ for my $am ('ao_row', 'ao_column')
+ {
+ my $table = "vstat_tail_$am";
+ $node->safe_psql('postgres', qq{
+CREATE TABLE $table (id int) USING $am;
+INSERT INTO $table SELECT generate_series(1, 10000);
+CHECKPOINT;
+});
+ my $relpath = $node->safe_psql('postgres',
+ "SELECT pg_relation_filepath('$table')");
+ my @files = grep { -f $_ && -s $_ }
+ glob($node->data_dir . '/' . $relpath . '*');
+ my ($file) = grep { /\Q$relpath\E(?:\.\d+)?$/ } @files;
+ defined($file) or BAIL_OUT("no data file for $table");
+ my $original_size = -s $file;
+ my $tail_size = 2**31 + 17;
+ # Model an aborted insert's tail in this disposable cluster with a
+ # sparse file. This exercises real truncate without writing 2 GiB.
+ open(my $fh, '+<', $file) or die "$file: $!";
+ truncate($fh, $original_size + $tail_size) or die "truncate: $!";
+ close($fh) or die "close: $!";
+ vacuum_table($table);
+ is(-s $file, $original_size, "$am removes the physical tail");
+ my $pages = int(($tail_size + $block_size - 1) / $block_size);
+ is(counters($table, 'bytes_removed, pages_deleted'), "$tail_size|$pages",
+ "$am preserves exact bytes and rounds the block equivalent up");
+ my $segs = counters($table, 'total_file_segs');
+ cmp_ok($segs, '>', 0, "$am has a segment snapshot");
+ $node->restart;
+ is(counters($table, 'bytes_removed, pages_deleted, total_file_segs'),
+ "$tail_size|$pages|$segs", "$am counters and state survive restart");
+ vacuum_table($table);
+ is(counters($table, 'bytes_removed, pages_deleted, total_file_segs'),
+ "$tail_size|$pages|$segs", "$am repeated vacuum does not recount the tail or segments");
+ run_and_wait("SELECT vacuum_stats_reset('$table'::regclass::oid)");
+ is(counters($table, 'bytes_removed, total_file_segs'), '0|0',
+ "$am relation reset clears both cumulative bytes and segment state");
+ }
+};
+
+subtest 'SP-GiST does not recount reusable pages' => sub {
+ for my $am ('heap', 'ao_row', 'ao_column')
+ {
+ my $table = "vstat_spg_$am";
+ $node->safe_psql('postgres', qq{
+CREATE TABLE $table (id int, p point) USING $am;
+CREATE INDEX ${table}_idx ON $table USING spgist (p);
+INSERT INTO $table SELECT i, point(i,i) FROM generate_series(1, 10000) g(i);
+DELETE FROM $table;
+});
+ vacuum_table($table, '(INDEX_CLEANUP ON)');
+ my $pages = index_counters("${table}_idx", 'pages_deleted');
+ is(index_counters("${table}_idx", 'tuples_deleted'), '10000',
+ "$am SP-GiST removes the original index entries");
+ vacuum_table($table, '(INDEX_CLEANUP ON)');
+ is(index_counters("${table}_idx", 'pages_deleted'), $pages,
+ "$am SP-GiST does not count already empty pages again");
+ is(index_counters("${table}_idx", 'bytes_removed'), '0',
+ "$am SP-GiST page reuse returns no bytes to the filesystem");
+ }
+};
+
+subtest 'VACUUM FULL leaves the extended counters unchanged' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_full (id int PRIMARY KEY) WITH (autovacuum_enabled = off);
+INSERT INTO vstat_full SELECT generate_series(1, 10000);
+DELETE FROM vstat_full WHERE id % 2 = 0;
+});
+ vacuum_table('vstat_full');
+ is(counters('vstat_full', 'tuples_deleted'), '5000',
+ 'plain vacuum establishes nonzero counters');
+ $node->safe_psql('postgres', 'DELETE FROM vstat_full');
+ wait_for_stats(
+ "SELECT n_tup_del = 10000 FROM pg_stat_all_tables_internal WHERE relname = 'vstat_full'",
+ 'DELETE report before VACUUM FULL');
+ my $table_before = counters('vstat_full');
+ my $index_before = index_counters('vstat_full_pkey');
+ run_and_wait('VACUUM FULL vstat_full');
+ is(counters('vstat_full'), $table_before, 'VACUUM FULL leaves table counters unchanged');
+ is(index_counters('vstat_full_pkey'), $index_before, 'VACUUM FULL leaves index counters unchanged');
+};
+
+subtest 'cost-based delay is part of total time' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_delay (id int) WITH (autovacuum_enabled = off);
+INSERT INTO vstat_delay SELECT generate_series(1, 10000);
+DELETE FROM vstat_delay;
+});
+ # Session-local settings do not leak into later scenarios.
+ my $verbose = vacuum_table('vstat_delay', 'VERBOSE',
+ 'SET track_cost_delay_timing = on; SET vacuum_cost_delay = 1; SET vacuum_cost_limit = 1;');
+ is(counters('vstat_delay', 'total_vacuum_delay_time > 0 AND total_vacuum_delay_time <= total_vacuum_time'),
+ 't', 'the measured cost delay is positive and included in total time');
+ my $delay = sprintf('%.3f', counters('vstat_delay', 'total_vacuum_delay_time'));
+ like($verbose, qr/elapsed: \d+\.\d{3} ms, cost-based delay: \Q$delay\E ms/,
+ 'VERBOSE reports elapsed time and the same measured cost delay');
+};
+
+subtest 'index pages are not counted again by later vacuums' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_idx (id int PRIMARY KEY) WITH (autovacuum_enabled = off);
+INSERT INTO vstat_idx SELECT generate_series(1, 100000);
+DELETE FROM vstat_idx WHERE id <= 90000;
+});
+ vacuum_table('vstat_idx');
+ my $pages = index_counters('vstat_idx_pkey', 'pages_deleted');
+ cmp_ok($pages, '>', 0, 'the first vacuum deleted index pages');
+ $node->safe_psql('postgres', 'DELETE FROM vstat_idx WHERE id > 99000');
+ vacuum_table('vstat_idx');
+ is(index_counters('vstat_idx_pkey', 'tuples_deleted'), '91000',
+ 'the next vacuum removed exactly 1000 more index entries');
+ is(index_counters('vstat_idx_pkey', "pages_deleted >= $pages AND pages_deleted - $pages < $pages / 2"),
+ 't', 'deleted-page totals accumulate without recounting the first run');
+};
+
+subtest 'reset preserves ordinary statistics and other relations' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_reset (id int PRIMARY KEY) WITH (autovacuum_enabled = off);
+CREATE TABLE vstat_keep (id int) WITH (autovacuum_enabled = off);
+INSERT INTO vstat_reset SELECT generate_series(1, 1000);
+INSERT INTO vstat_keep SELECT generate_series(1, 1000);
+DELETE FROM vstat_keep;
+});
+ vacuum_table('vstat_keep');
+ vacuum_table('vstat_reset', 'FREEZE');
+ $node->safe_psql('postgres', 'UPDATE vstat_reset SET id = id + 1000');
+ wait_for_updates('vstat_reset', 1000);
+ vacuum_table('vstat_reset', 'FREEZE');
+ is(counters('vstat_reset',
+ 'tuples_deleted, frozen_page_marks_cleared > 0, visible_page_marks_cleared > 0'),
+ '1000|t|t', 'populate removal and VM revision counters before reset');
+ my $keep_before = counters('vstat_keep');
+ my $index_before = index_counters('vstat_reset_pkey');
+ my $db_before = database_counters();
+ my $ordinary_query = "SELECT vacuum_count, n_tup_upd FROM pg_stat_all_tables_internal WHERE relname = 'vstat_reset'";
+ my $ordinary_before = $node->safe_psql('postgres', $ordinary_query);
+ my $reset_before = counters('vstat_reset');
+
+ my ($result, $stdout, $stderr) = $node->psql('postgres',
+ 'SELECT vacuum_stats_reset(0::oid)');
+ is($result, 3, 'an invalid relation OID is rejected');
+ like($stderr, qr/invalid relation OID: 0/, 'invalid OID has a specific error');
+ run_and_wait('SELECT vacuum_stats_reset(NULL::oid)');
+ is(counters('vstat_reset'), $reset_before,
+ 'invalid and NULL OIDs do not reset a relation');
+ is(database_counters(), $db_before,
+ 'invalid and NULL OIDs do not reset database totals');
+
+ run_and_wait("SELECT vacuum_stats_reset('vstat_reset'::regclass::oid)");
+ is(counters('vstat_reset', $all_zero), 't', 'relation reset clears every vacuum counter including VM clearing counters');
+ is(counters('vstat_keep'), $keep_before, 'relation reset preserves another table');
+ is(index_counters('vstat_reset_pkey'), $index_before, 'relation reset preserves the index counters');
+ is(database_counters(), $db_before, 'relation reset preserves database totals');
+ is($node->safe_psql('postgres', $ordinary_query), $ordinary_before,
+ 'relation reset preserves ordinary statistics');
+
+ # Refill the reset relation, including both VM revision counters,
+ # before checking the database-wide reset.
+ $node->safe_psql('postgres', 'UPDATE vstat_reset SET id = id + 1000');
+ wait_for_updates('vstat_reset', 2000);
+ vacuum_table('vstat_reset', 'FREEZE');
+ is(counters('vstat_reset',
+ 'tuples_deleted, frozen_page_marks_cleared > 0, visible_page_marks_cleared > 0'),
+ '1000|t|t', 'the counters are nonzero again before database reset');
+ $ordinary_before = $node->safe_psql('postgres', $ordinary_query);
+ run_and_wait('SELECT vacuum_stats_reset()');
+ is($node->safe_psql('postgres', "SELECT bool_and($all_zero) FROM pg_stat_vacuum_tables"),
+ 't', 'database reset clears all table counters');
+ is($node->safe_psql('postgres', 'SELECT bool_and(total_file_segs = 0) FROM pg_stat_vacuum_tables'),
+ 't', 'database reset clears AO segment snapshots');
+ is($node->safe_psql('postgres', "SELECT bool_and($all_zero) FROM pg_stat_vacuum_indexes"),
+ 't', 'database reset clears all index counters');
+ is(database_counters($all_zero), 't', 'database reset clears database totals');
+ is($node->safe_psql('postgres', $ordinary_query), $ordinary_before,
+ 'database reset preserves ordinary statistics');
+
+
+};
+
+subtest 'database totals do not count index work twice' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_db (id int PRIMARY KEY, val int) WITH (autovacuum_enabled = off);
+CREATE INDEX ON vstat_db (val);
+INSERT INTO vstat_db SELECT i, i FROM generate_series(1, 10000) g(i);
+DELETE FROM vstat_db WHERE id % 2 = 0;
+});
+ run_and_wait('SELECT vacuum_stats_reset()');
+ vacuum_table('vstat_db');
+ is($node->safe_psql('postgres',
+ "SELECT sum(tuples_deleted) FROM pg_stat_vacuum_indexes WHERE relname = 'vstat_db'"),
+ '10000', 'both indexes report 5000 removed entries');
+ is(database_counters('tuples_deleted'), '5000', 'database totals count only heap tuples');
+ is(database_counters(), counters('vstat_db'),
+ 'all database vacuum counters match the only table vacuumed since reset');
+};
+
+subtest 'statistics snapshots release their vacuum counters' => sub {
+ # Keep the last snapshot alive until the context count is read.
+ # Outside this transaction its context would already have been freed.
+ is($node->safe_psql('postgres', q{
+BEGIN;
+DO $$
+BEGIN
+ FOR i IN 1..50 LOOP
+ PERFORM pg_stat_clear_snapshot();
+ PERFORM count(*) FROM pg_stat_vacuum_tables WHERE tuples_deleted > 0;
+ END LOOP;
+END
+$$;
+SELECT count(*) FROM pg_backend_memory_contexts
+WHERE name = 'Databases hash';
+COMMIT;
+}), '1', 'only the current snapshot owns statistics entries with vacuum counters');
+};
+
+subtest 'reset targets shared catalogs independently' => sub {
+ # Shared catalogs have a separate collector entry, not the current DB's.
+ # Create actual dead index entries instead of timing an empty cleanup.
+ $node->safe_psql('postgres', 'CREATE DATABASE vstat_shared_reset');
+ $node->safe_psql('postgres', 'DROP DATABASE vstat_shared_reset');
+ vacuum_table('pg_database', '(FREEZE, INDEX_CLEANUP ON)');
+ is(counters('pg_database', 'total_vacuum_time > 0'), 't',
+ 'populate vacuum counters for a shared catalog');
+ my $shared_before = counters('pg_database');
+ run_and_wait('SELECT vacuum_stats_reset()');
+ is(counters('pg_database'), $shared_before,
+ 'database reset leaves shared catalog counters alone');
+
+ vacuum_table('vstat_reset');
+ my $reset_before = counters('vstat_reset');
+ my $db_before = database_counters();
+ run_and_wait("SELECT vacuum_stats_reset('pg_database'::regclass::oid)");
+ is(counters('pg_database', $all_zero), 't',
+ 'relation reset reaches the shared catalog entry');
+ is(counters('vstat_reset'), $reset_before,
+ 'shared relation reset leaves current-database relations alone');
+ is(database_counters(), $db_before,
+ 'shared relation reset leaves current-database totals alone');
+
+ # Indexes are separate reset targets, including shared catalog indexes.
+ is(index_counters('pg_database_oid_index', 'tuples_deleted > 0'), 't',
+ 'resetting the shared table preserves its index counters');
+ run_and_wait(q{
+SELECT vacuum_stats_reset(indexrelid) FROM pg_index
+WHERE indrelid = 'pg_database'::regclass
+});
+ is(index_counters('pg_database_oid_index', $all_zero), 't',
+ 'relation reset also reaches shared index counters');
+};
+
+subtest 'disabling tracking omits vacuum counters and preserves ordinary statistics' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_startup (id int PRIMARY KEY) WITH (autovacuum_enabled = off);
+INSERT INTO vstat_startup SELECT generate_series(1, 1000);
+});
+ vacuum_table('vstat_startup', 'FREEZE');
+ $node->safe_psql('postgres', 'UPDATE vstat_startup SET id = id + 1000');
+ wait_for_updates('vstat_startup', 1000);
+ vacuum_table('vstat_startup', 'FREEZE');
+ is(counters('vstat_startup',
+ 'tuples_deleted, frozen_page_marks_cleared > 0, visible_page_marks_cleared > 0'),
+ '1000|t|t', 'populate vacuum and VM revision counters before disabling');
+ my $ordinary_query = "SELECT vacuum_count, n_tup_upd FROM pg_stat_all_tables_internal WHERE relname = 'vstat_startup'";
+ my $ordinary_before = $node->safe_psql('postgres', $ordinary_query);
+ my $vm_before = counters('vstat_startup', $vm_columns);
+ my $timing_columns = 'total_vacuum_time, total_autovacuum_time, total_vacuum_delay_time, total_autovacuum_delay_time';
+ my $timing_before = counters('vstat_startup', $timing_columns);
+ my $db_vm_before = database_counters($vm_columns);
+ my $on_bytes = snapshot_bytes();
+ $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = off');
+ $node->restart;
+ is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'off',
+ 'restart disables tracking');
+ is(counters('vstat_startup', $vm_columns), $vm_before,
+ 'disabling vacuum tracking preserves relation VM counters');
+ is(counters('vstat_startup', $timing_columns), $timing_before,
+ 'disabling extended tracking preserves core vacuum timing');
+ is(database_counters($vm_columns), $db_vm_before,
+ 'disabling vacuum tracking preserves database VM totals');
+ is(counters('vstat_startup', $vacuum_zero), 't', 'disabled table counters read as zero');
+ is(index_counters('vstat_startup_pkey', $vacuum_zero), 't', 'disabled index counters read as zero');
+ is(database_counters($vacuum_zero), 't', 'disabled database counters read as zero');
+ is($node->safe_psql('postgres', 'SELECT bool_and(total_file_segs = 0) FROM pg_stat_vacuum_tables'),
+ 't', 'disabled AO segment snapshots read as zero');
+ is($node->safe_psql('postgres', $ordinary_query), $ordinary_before,
+ 'reading statistics with tracking disabled preserves ordinary counters');
+ cmp_ok(snapshot_bytes(), '<', $on_bytes,
+ 'disabled snapshots omit vacuum storage but retain ordinary VM counters');
+ run_and_wait("SELECT vacuum_stats_reset('vstat_startup'::regclass::oid)");
+ is(counters('vstat_startup', $vm_columns), '0|0',
+ 'relation reset clears VM counters with vacuum tracking off');
+ is(database_counters($vm_columns), $db_vm_before,
+ 'relation reset preserves database VM totals with vacuum tracking off');
+ run_and_wait('SELECT vacuum_stats_reset()');
+ is(database_counters($vm_columns), '0|0',
+ 'database reset clears VM totals with vacuum tracking off');
+ is($node->safe_psql('postgres', $ordinary_query), $ordinary_before,
+ 'vacuum reset while disabled leaves ordinary counters intact');
+ $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = on');
+ $node->restart;
+ is(counters('vstat_startup', $vacuum_zero), 't',
+ 're-enabling does not restore discarded table vacuum counters');
+ is(database_counters($vacuum_zero), 't',
+ 're-enabling does not restore discarded database vacuum counters');
+ is($node->safe_psql('postgres', $ordinary_query), $ordinary_before,
+ 'ordinary statistics survive both changes in record size');
+ vacuum_table('vstat_startup', 'FREEZE');
+ is(counters('vstat_startup', 'total_vacuum_time > 0'), 't',
+ 'vacuum reporting resumes after tracking is enabled again');
+};
+
+subtest 'clean restart preserves statistics; crash recovery resets them' => sub {
+ $node->safe_psql('postgres', q{
+CREATE TABLE vstat_restart (id int PRIMARY KEY) WITH (autovacuum_enabled = off);
+INSERT INTO vstat_restart SELECT generate_series(1, 1000);
+DELETE FROM vstat_restart WHERE id % 2 = 0;
+});
+ vacuum_table('vstat_restart', 'FREEZE');
+ is(counters('vstat_restart', 'tuples_deleted'), '500', 'table counters are populated before restart');
+ is(index_counters('vstat_restart_pkey', 'tuples_deleted'), '500', 'index counters are populated before restart');
+ my $table_before = counters('vstat_restart');
+ my $index_before = index_counters('vstat_restart_pkey');
+ my $db_before = database_counters();
+ $node->restart;
+ is(counters('vstat_restart'), $table_before, 'table counters survive a clean restart');
+ is(index_counters('vstat_restart_pkey'), $index_before, 'index counters survive a clean restart');
+ is(database_counters(), $db_before, 'database counters survive a clean restart');
+ $node->stop('immediate');
+ $node->start;
+ is(counters('vstat_restart', $all_zero), 't', 'crash recovery resets table counters');
+ is(index_counters('vstat_restart_pkey', $all_zero), 't', 'crash recovery resets index counters');
+ is(database_counters($all_zero), 't', 'crash recovery resets database counters');
+};
+
+$node->stop;
+done_testing();
diff --git a/contrib/vacuum_stats/t/002_index_vacuum_time.pl b/contrib/vacuum_stats/t/002_index_vacuum_time.pl
new file mode 100644
index 00000000000..ff41161e7fa
--- /dev/null
+++ b/contrib/vacuum_stats/t/002_index_vacuum_time.pl
@@ -0,0 +1,252 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+
+# Test the total_vacuum_time counter of pg_stat_vacuum_indexes: the time vacuum
+# spent processing each index, accumulated over bulkdelete and cleanup passes,
+# mirroring the table-level counter in pg_stat_vacuum_tables.
+
+use strict;
+use warnings FATAL => 'all';
+use PostgresNode;
+use TestLib;
+use Test::More;
+
+my $node = get_new_node('main');
+$node->init;
+$node->append_conf(
+ 'postgresql.conf', qq[
+autovacuum = off
+track_cost_delay_timing = on
+]);
+$node->start;
+$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats');
+
+# The index has to span enough pages for its vacuum passes to exceed
+# vacuum_cost_limit on buffer hits alone (its pages are still dirty from the
+# load, so dirtying them costs nothing). Scale the row count with the block
+# size, so that builds with larger blocks (32kB in Cloudberry) keep the page
+# count of the default 8kB.
+my $nrows = 100000 *
+ ($node->safe_psql('postgres', 'SHOW block_size') / 8192);
+
+# Enough rows that the bulkdelete pass over the index takes measurable time;
+# a small vacuum_cost_delay adds deterministic delay time on fast machines.
+$node->safe_psql(
+ 'postgres', qq[
+CREATE TABLE vactime_t (id int PRIMARY KEY, v text) WITH (autovacuum_enabled = off);
+INSERT INTO vactime_t SELECT g, repeat('x', 10) FROM generate_series(1, $nrows) g;
+DELETE FROM vactime_t WHERE id % 2 = 0;
+]);
+$node->safe_psql(
+ 'postgres', qq[
+SET vacuum_cost_delay = '1ms';
+SET vacuum_cost_limit = 200;
+VACUUM vactime_t;
+]);
+
+# The collector receives the ordinary vacuum and index-pass reports asynchronously.
+$node->poll_query_until('postgres', q{
+SELECT total_vacuum_time > 0 AND total_vacuum_delay_time > 0
+FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey'
+}) or BAIL_OUT('index timing report did not reach the collector');
+$node->poll_query_until('postgres', q{
+SELECT total_vacuum_time > 0 AND total_vacuum_delay_time > 0
+FROM pg_stat_vacuum_database WHERE datname = current_database()
+}) or BAIL_OUT('database timing report did not reach the collector');
+
+# The upper bound guards against garbage such as an epoch-based elapsed time
+# leaking into the counter.
+is( $node->safe_psql(
+ 'postgres', qq[
+SELECT total_vacuum_time > 0 AND total_vacuum_time < 600000
+ FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey']),
+ 't',
+ 'total_vacuum_time advanced sanely for the index in pg_stat_vacuum_indexes');
+
+is( $node->safe_psql(
+ 'postgres', qq[
+SELECT total_vacuum_delay_time > 0 AND total_vacuum_delay_time <= total_vacuum_time
+ FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey']),
+ 't',
+ 'total_vacuum_delay_time advanced and not above total_vacuum_time');
+
+is( $node->safe_psql(
+ 'postgres', qq[
+SELECT total_autovacuum_time = 0 FROM pg_stat_vacuum_indexes
+ WHERE indexrelname = 'vactime_t_pkey']),
+ 't',
+ 'manual vacuum did not count into total_autovacuum_time');
+
+# The same run must have accumulated into the database-wide totals.
+is( $node->safe_psql(
+ 'postgres', qq[
+SELECT total_vacuum_time > 0 AND total_vacuum_time < 600000
+ AND total_vacuum_delay_time > 0
+ AND total_vacuum_delay_time <= total_vacuum_time
+ FROM pg_stat_vacuum_database WHERE datname = current_database()]),
+ 't',
+ 'database-wide vacuum times advanced sanely in pg_stat_vacuum_database');
+
+# The database total is the sum of table runs, which already include index
+# work. Adding the positive per-index time again would break this equality.
+# Cast each millisecond value to numeric before summing to avoid floating-point
+# rounding differences. Shared relations belong to the database with OID zero.
+is($node->safe_psql('postgres', q{
+SELECT d.total_vacuum_time::numeric = t.elapsed
+ AND d.total_vacuum_delay_time::numeric = t.delay
+FROM pg_stat_vacuum_database d CROSS JOIN (
+ SELECT sum(s.total_vacuum_time::numeric) AS elapsed,
+ sum(s.total_vacuum_delay_time::numeric) AS delay
+ FROM pg_stat_vacuum_tables s JOIN pg_class c ON c.oid = s.relid
+ WHERE NOT c.relisshared
+) t
+WHERE d.datname = current_database()
+}), 't', 'database totals include index work exactly once');
+
+# Turning off delay timing must preserve the accumulated delay while elapsed
+# time keeps advancing. Keep cost delays enabled to exercise the timing GUC.
+my $manual_times = q{
+SELECT total_vacuum_time, total_vacuum_delay_time
+FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey'
+};
+my ($elapsed_before, $delay_before) =
+ split /\|/, $node->safe_psql('postgres', $manual_times);
+$node->safe_psql('postgres', q{
+SET track_cost_delay_timing = off;
+SET vacuum_cost_delay = '1ms';
+SET vacuum_cost_limit = 200;
+DELETE FROM vactime_t WHERE id % 4 = 1;
+VACUUM (INDEX_CLEANUP ON) vactime_t;
+});
+$node->poll_query_until('postgres',
+ "SELECT total_vacuum_time > $elapsed_before FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey'")
+ or BAIL_OUT('index elapsed time did not advance with delay timing disabled');
+my ($elapsed_after, $delay_after) =
+ split /\|/, $node->safe_psql('postgres', $manual_times);
+is($delay_after, $delay_before, 'disabled delay timing preserves the index delay total');
+
+# Exercise ANALYZE and VACUUM in the same backend, where the process-local
+# delay accumulator survives between statements. Check their SQL totals
+# independently rather than inferring one from the other.
+# Drain earlier asynchronous vacuum reports before checking what ANALYZE
+# alone changes.
+$node->restart;
+my $table_manual_times = q{
+SELECT total_vacuum_time, total_vacuum_delay_time
+FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t'
+};
+my $db_manual_times = q{
+SELECT total_vacuum_time, total_vacuum_delay_time
+FROM pg_stat_vacuum_database WHERE datname = current_database()
+};
+my @timing_queries = ($table_manual_times, $manual_times, $db_manual_times);
+my @timing_labels = ('table', 'index', 'database');
+my @before_analyze = map { $node->safe_psql('postgres', $_) } @timing_queries;
+$node->safe_psql('postgres', q{
+SET vacuum_cost_delay = '1ms';
+SET vacuum_cost_limit = 200;
+ANALYZE vactime_t;
+});
+$node->poll_query_until('postgres', q{
+SELECT total_analyze_time > 0 FROM pg_stat_vacuum_tables
+WHERE relname = 'vactime_t'
+}) or BAIL_OUT('ANALYZE timing report did not reach the collector');
+for my $i (0..2)
+{
+ is($node->safe_psql('postgres', $timing_queries[$i]), $before_analyze[$i],
+ "ANALYZE preserves $timing_labels[$i] vacuum elapsed and delay times");
+}
+my $analyze_before = $node->safe_psql('postgres', q{
+SELECT total_analyze_time FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t'
+});
+my ($table_elapsed, $table_delay) = split /\|/, $before_analyze[0];
+$node->safe_psql('postgres', q{
+SET vacuum_cost_delay = '1ms';
+SET vacuum_cost_limit = 200;
+ANALYZE vactime_t;
+VACUUM (ANALYZE, DISABLE_PAGE_SKIPPING, INDEX_CLEANUP ON) vactime_t;
+ANALYZE vactime_t;
+});
+$node->poll_query_until('postgres', qq{
+SELECT total_vacuum_time > $table_elapsed
+ AND total_vacuum_delay_time > $table_delay
+ AND total_analyze_time > $analyze_before
+FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t'
+}) or BAIL_OUT('VACUUM ANALYZE did not report both maintenance phases');
+is($node->safe_psql('postgres', q{
+SELECT total_vacuum_delay_time <= total_vacuum_time
+ AND total_autovacuum_time = 0 AND total_autoanalyze_time = 0
+FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t'
+}), 't', 'VACUUM ANALYZE records separate manual timing totals');
+
+# Cloudberry autovacuums catalogs. Churn comments to create dead catalog
+# tuples and index entries, then wait for a real worker to process them.
+$node->safe_psql('postgres', q{
+CREATE TABLE vactime_autovacuum (id int) WITH (autovacuum_enabled = off);
+COMMENT ON TABLE vactime_autovacuum IS 'initial';
+VACUUM (INDEX_CLEANUP ON) pg_description;
+});
+$node->poll_query_until('postgres', q{
+SELECT total_vacuum_time > 0 FROM pg_stat_vacuum_tables WHERE relname = 'pg_description'
+}) or BAIL_OUT('catalog vacuum timing did not reach the collector');
+my $table_time = "SELECT total_vacuum_time, total_autovacuum_time FROM pg_stat_vacuum_tables WHERE relname = 'pg_description'";
+my $index_time = "SELECT total_vacuum_time, total_autovacuum_time FROM pg_stat_vacuum_indexes WHERE indexrelname = 'pg_description_o_c_o_index'";
+my $db_time = "SELECT total_vacuum_time, total_autovacuum_time FROM pg_stat_vacuum_database WHERE datname = current_database()";
+my @manual_before = map { (split /\|/, $node->safe_psql('postgres', $_))[0] }
+ ($table_time, $index_time, $db_time);
+$node->safe_psql('postgres', q{
+DO $$ BEGIN
+ FOR i IN 1..200 LOOP
+ EXECUTE 'COMMENT ON TABLE vactime_autovacuum IS NULL';
+ EXECUTE format('COMMENT ON TABLE vactime_autovacuum IS %L', i::text);
+ END LOOP;
+END $$;
+ALTER SYSTEM SET autovacuum_naptime = '1s';
+ALTER SYSTEM SET autovacuum_vacuum_threshold = 0;
+ALTER SYSTEM SET autovacuum_vacuum_scale_factor = 0;
+ALTER SYSTEM SET autovacuum_vacuum_insert_threshold = -1;
+ALTER SYSTEM SET autovacuum = on;
+});
+$node->reload;
+$node->poll_query_until('postgres', q{
+SELECT total_autovacuum_time > 0 FROM pg_stat_vacuum_tables WHERE relname = 'pg_description'
+}) or BAIL_OUT('autovacuum did not report catalog time');
+$node->poll_query_until('postgres', q{
+SELECT total_autovacuum_time > 0 FROM pg_stat_vacuum_indexes
+WHERE indexrelname = 'pg_description_o_c_o_index'
+}) or BAIL_OUT('autovacuum did not report catalog index time');
+$node->safe_psql('postgres', 'ALTER SYSTEM SET autovacuum = off');
+# Drain the collector on a clean shutdown before comparing exact totals.
+# A worker disappearing from pg_stat_activity alone does not prove that
+# its final UDP reports have reached the collector.
+$node->restart;
+my @auto_before;
+my @queries = ($table_time, $index_time, $db_time);
+my @labels = ('table', 'index', 'database');
+for my $i (0..2)
+{
+ my ($manual, $automatic) = split /\|/, $node->safe_psql('postgres', $queries[$i]);
+ is($manual, $manual_before[$i], "autovacuum preserves manual $labels[$i] time");
+ cmp_ok($automatic, '>', 0, "autovacuum records separate $labels[$i] time");
+ push @auto_before, $automatic;
+}
+$node->safe_psql('postgres', 'VACUUM (INDEX_CLEANUP ON) pg_description');
+$node->poll_query_until('postgres',
+ "SELECT total_vacuum_time > $manual_before[0] FROM pg_stat_vacuum_tables WHERE relname = 'pg_description'")
+ or BAIL_OUT('manual catalog vacuum did not report new time');
+for my $i (0..2)
+{
+ my ($manual, $automatic) = split /\|/, $node->safe_psql('postgres', $queries[$i]);
+ cmp_ok($manual, '>', $manual_before[$i], "manual vacuum adds $labels[$i] time");
+ is($automatic, $auto_before[$i], "manual vacuum preserves autovacuum $labels[$i] time");
+}
+my @before_restart = map { $node->safe_psql('postgres', $_) } @queries;
+$node->restart;
+for my $i (0..2)
+{
+ is($node->safe_psql('postgres', $queries[$i]), $before_restart[$i],
+ "both $labels[$i] timing counters survive restart");
+}
+
+$node->stop;
+
+done_testing();
diff --git a/contrib/vacuum_stats/t/003_vacuum_failsafe.pl b/contrib/vacuum_stats/t/003_vacuum_failsafe.pl
new file mode 100644
index 00000000000..a41cd054b53
--- /dev/null
+++ b/contrib/vacuum_stats/t/003_vacuum_failsafe.pl
@@ -0,0 +1,113 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+
+# Test that vacuums entering the wraparound failsafe mode are counted in
+# pg_stat_vacuum_tables.vacuum_failsafe_count and aggregated per database in
+# pg_stat_vacuum_database.vacuum_failsafe_count.
+use strict;
+use warnings FATAL => 'all';
+use PostgresNode;
+use TestLib;
+use Test::More;
+
+my $node = get_new_node('main');
+$node->init;
+# The failsafe cutoff is clamped to 1.05 * autovacuum_freeze_max_age, so use
+# the minimum allowed value to keep the number of XIDs to burn small.
+$node->append_conf(
+ 'postgresql.conf', qq[
+autovacuum = off
+autovacuum_freeze_max_age = 100000
+]);
+$node->start;
+$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats');
+
+$node->safe_psql(
+ 'postgres', qq[
+ CREATE TABLE tab_failsafe (i int);
+ INSERT INTO tab_failsafe SELECT generate_series(1, 100);
+]);
+
+# A vacuum without failsafe pressure must not bump the counter.
+$node->safe_psql('postgres', 'VACUUM FREEZE tab_failsafe;');
+$node->poll_query_until('postgres', q{
+SELECT vacuum_count = 1 FROM pg_stat_all_tables_internal WHERE relname = 'tab_failsafe'
+}) or BAIL_OUT('normal vacuum report did not reach the collector');
+my $count = $node->safe_psql('postgres',
+ q[SELECT vacuum_failsafe_count FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe';]
+);
+is($count, '0', 'aggressive VACUUM FREEZE does not count as failsafe');
+
+# Age the table past 1.05 * autovacuum_freeze_max_age: burn XIDs with
+# aborted subtransactions (each aborted subxact consumes an assigned XID).
+$node->safe_psql(
+ 'postgres', qq[
+ CREATE TABLE burn_xids (i int);
+ DO \$\$
+ BEGIN
+ FOR i IN 1..110000 LOOP
+ BEGIN
+ INSERT INTO burn_xids VALUES (1);
+ RAISE EXCEPTION 'burn';
+ EXCEPTION WHEN OTHERS THEN
+ END;
+ END LOOP;
+ END \$\$;
+]);
+
+# vacuum_failsafe_age = 0 makes the (clamped) cutoff kick in immediately.
+$node->safe_psql(
+ 'postgres', qq[
+ SET vacuum_failsafe_age = 0;
+ SET vacuum_multixact_failsafe_age = 0;
+ VACUUM tab_failsafe;
+]);
+
+$node->poll_query_until('postgres', q{
+SELECT vacuum_failsafe_count = 1 FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe'
+}) or BAIL_OUT('failsafe vacuum report did not reach the collector');
+
+$count = $node->safe_psql('postgres',
+ q[SELECT vacuum_failsafe_count FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe';]
+);
+is($count, '1', 'failsafe vacuum counted in pg_stat_vacuum_tables');
+
+my $db_count = $node->safe_psql('postgres',
+ q[SELECT vacuum_failsafe_count FROM pg_stat_vacuum_database WHERE datname = 'postgres';]
+);
+is($db_count, '1', 'failsafe vacuum counted in pg_stat_vacuum_database');
+
+# Once the table has been frozen, another vacuum must not count the same
+# failsafe event again. Wait for that run's report, not for an unchanged value.
+$node->safe_psql('postgres', 'VACUUM tab_failsafe');
+$node->poll_query_until('postgres', q{
+SELECT vacuum_count = 3 FROM pg_stat_all_tables_internal WHERE relname = 'tab_failsafe'
+}) or BAIL_OUT('subsequent vacuum report did not reach the collector');
+is($node->safe_psql('postgres', q{
+SELECT t.vacuum_failsafe_count, d.vacuum_failsafe_count
+FROM pg_stat_vacuum_tables t CROSS JOIN pg_stat_vacuum_database d
+WHERE t.relname = 'tab_failsafe' AND d.datname = current_database()
+}), '1|1', 'subsequent ordinary vacuum does not count the failsafe again');
+
+$node->restart;
+is($node->safe_psql('postgres', q{
+SELECT t.vacuum_failsafe_count, d.vacuum_failsafe_count
+FROM pg_stat_vacuum_tables t CROSS JOIN pg_stat_vacuum_database d
+WHERE t.relname = 'tab_failsafe' AND d.datname = current_database()
+}), '1|1', 'relation and database failsafe counts survive a clean restart');
+
+$node->safe_psql('postgres',
+ q{SELECT pg_stat_reset_single_table_counters('tab_failsafe'::regclass)});
+$node->poll_query_until('postgres', q{
+SELECT vacuum_failsafe_count = 0 FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe'
+}) or BAIL_OUT('relation failsafe counter was not reset');
+is($node->safe_psql('postgres', q{
+SELECT vacuum_failsafe_count FROM pg_stat_vacuum_database WHERE datname = current_database()
+}), '1', 'resetting a relation preserves the database failsafe total');
+
+$node->safe_psql('postgres', 'SELECT pg_stat_reset()');
+ok($node->poll_query_until('postgres', q{
+SELECT vacuum_failsafe_count = 0 FROM pg_stat_vacuum_database WHERE datname = current_database()
+}), 'database reset clears the failsafe total');
+
+$node->stop;
+done_testing();
diff --git a/contrib/vacuum_stats/t/004_visibility_map_stats.pl b/contrib/vacuum_stats/t/004_visibility_map_stats.pl
new file mode 100644
index 00000000000..66256234b09
--- /dev/null
+++ b/contrib/vacuum_stats/t/004_visibility_map_stats.pl
@@ -0,0 +1,216 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+
+# VM flag clearings are ordinary relation/database statistics. This test
+# reads them through the vacuum_stats extension without changing core views.
+# Adapt the v44 VM stability scenario to the collector used by Cloudberry:
+# https://www.postgresql.org/message-id/attachment/204710/v44-0009-Track-table-VM-stability.patch
+use strict;
+use warnings FATAL => 'all';
+use PostgresNode;
+use TestLib;
+use Test::More;
+
+my $node = get_new_node('vm_stats');
+$node->init;
+$node->append_conf('postgresql.conf', q{
+autovacuum = off
+track_counts = on
+});
+$node->start;
+$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats');
+
+sub wait_for_stats
+{
+ my ($sql, $description) = @_;
+ $node->poll_query_until('postgres', $sql)
+ or BAIL_OUT("timed out waiting for $description");
+}
+
+my $columns = 'frozen_page_marks_cleared, visible_page_marks_cleared';
+my $table_query = "SELECT $columns FROM pg_stat_vacuum_tables WHERE relname = 'vm_heap'";
+my $visible_query = "SELECT $columns FROM pg_stat_vacuum_tables WHERE relname = 'vm_visible'";
+my $db_query = "SELECT $columns FROM pg_stat_vacuum_database WHERE datname = current_database()";
+
+sub populated_pages
+{
+ my ($table) = @_;
+ return $node->safe_psql('postgres',
+ "SELECT count(DISTINCT split_part(ctid::text, ',', 1)) FROM $table");
+}
+
+$node->safe_psql('postgres', q{
+CREATE TABLE vm_heap (id int PRIMARY KEY) WITH (fillfactor = 70);
+INSERT INTO vm_heap SELECT generate_series(1, 1000);
+CREATE TABLE vm_visible (id int);
+INSERT INTO vm_visible SELECT generate_series(1, 1000);
+});
+wait_for_stats(q{
+SELECT n_tup_ins = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap'
+}, 'insert report');
+$node->safe_psql('postgres', 'VACUUM FREEZE vm_heap');
+wait_for_stats(q{
+SELECT vacuum_count = 1 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap'
+}, 'vacuum report');
+# A normal vacuum marks the new tuples visible, without freezing their XIDs.
+$node->safe_psql('postgres', q{
+SET vacuum_freeze_min_age = 1000000000;
+SET vacuum_freeze_table_age = 1000000000;
+VACUUM vm_visible;
+});
+wait_for_stats(q{
+SELECT vacuum_count = 1 AND n_tup_ins = 1000
+FROM pg_stat_all_tables_internal WHERE relname = 'vm_visible'
+}, 'non-freezing vacuum and insert reports');
+
+# Drain reports from table/index creation before taking the database baseline.
+$node->restart;
+is($node->safe_psql('postgres', $table_query), '0|0',
+ 'setting VM flags does not count as clearing them');
+my ($db_frozen, $db_visible) = split /\|/,
+ $node->safe_psql('postgres', $db_query);
+my $pages = populated_pages('vm_heap');
+cmp_ok($pages, '>', 0, 'test table has populated heap pages');
+
+# Count each physical transition once, even though UPDATE touches many tuples
+# on each page. DML and VM counters travel in the same collector message, so
+# wait for the DML report before checking exact values, including zeroes.
+$node->safe_psql('postgres', 'UPDATE vm_heap SET id = id + 1000');
+wait_for_stats(q{
+SELECT n_tup_upd = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap'
+}, 'first update report');
+is($node->safe_psql('postgres', $table_query), "$pages|$pages",
+ 'UPDATE clears one all-visible and all-frozen mark per populated page');
+is($node->safe_psql('postgres', qq{
+SELECT frozen_page_marks_cleared - $db_frozen,
+ visible_page_marks_cleared - $db_visible
+FROM pg_stat_vacuum_database WHERE datname = current_database()
+}), "$pages|$pages", 'database totals include exactly the table transitions');
+
+$node->safe_psql('postgres', 'UPDATE vm_heap SET id = id + 1000');
+wait_for_stats(q{
+SELECT n_tup_upd = 2000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap'
+}, 'second update report');
+is($node->safe_psql('postgres', $table_query), "$pages|$pages",
+ 'updates on pages whose marks are already clear add nothing');
+is($node->safe_psql('postgres', qq{
+SELECT pg_stat_get_frozen_page_marks_cleared('vm_heap'::regclass),
+ pg_stat_get_visible_page_marks_cleared('vm_heap'::regclass)
+}), "$pages|$pages", 'extension getters and table view expose the same counters');
+
+# Distinguish the two counters and verify that rolling back DML does not undo
+# the physical clearing of VM bits. Utility-mode SELECT FOR UPDATE takes a
+# table lock, so use unfrozen tuples to exercise independent bit accounting.
+my $visible_pages = populated_pages('vm_visible');
+cmp_ok($visible_pages, '>', 0, 'unfrozen table has populated heap pages');
+$node->safe_psql('postgres', 'BEGIN; DELETE FROM vm_visible; ROLLBACK;');
+wait_for_stats(q{
+SELECT n_tup_del = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_visible'
+}, 'aborted delete report');
+is($node->safe_psql('postgres', 'SELECT count(*) FROM vm_visible'), '1000',
+ 'aborted DELETE preserves the rows');
+is($node->safe_psql('postgres', $visible_query), "0|$visible_pages",
+ 'aborted DELETE counts all-visible clearings without inventing all-frozen clearings');
+is($node->safe_psql('postgres', qq{
+SELECT frozen_page_marks_cleared - $db_frozen,
+ visible_page_marks_cleared - $db_visible
+FROM pg_stat_vacuum_database WHERE datname = current_database()
+}), $pages . '|' . ($pages + $visible_pages),
+ 'database totals distinguish the bits and include aborted DML');
+
+# Re-establish flags, then modify the table while a reader holds both an MVCC
+# snapshot and a statistics snapshot. The statistics snapshot postpones
+# observing the clearings; the VM bits themselves are cleared by DELETE,
+# before the reader commits.
+$node->safe_psql('postgres', 'VACUUM FREEZE vm_heap');
+wait_for_stats(q{
+SELECT vacuum_count = 2 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap'
+}, 'second vacuum report');
+is($node->safe_psql('postgres', $table_query), "$pages|$pages",
+ 'restoring VM flags does not increase clearing counters');
+my $refrozen_pages = populated_pages('vm_heap');
+my $cleared = $pages + $refrozen_pages;
+
+my ($in, $out) = ('', '');
+my $timer = IPC::Run::timeout($TestLib::timeout_default);
+my $reader = $node->background_psql('postgres', \$in, \$out, $timer);
+my $reader_query = sub {
+ my ($sql) = @_;
+ $out = '';
+ $in = "$sql;\n\\echo vm_query_done\n";
+ pump_until($reader, $timer, \$out, qr/^vm_query_done\r?$/m)
+ or BAIL_OUT('reader did not complete its query');
+ $out =~ s/\r//g;
+ $out =~ s/^vm_query_done\n?//m;
+ $out =~ s/^\n+|\n+$//g;
+ return $out;
+};
+is($reader_query->("BEGIN ISOLATION LEVEL REPEATABLE READ; $table_query"),
+ "$pages|$pages", 'reader caches the pre-delete statistics');
+is($reader_query->('SELECT count(*) FROM vm_heap'), '1000',
+ 'reader holds an MVCC snapshot of the rows');
+
+$node->safe_psql('postgres', 'DELETE FROM vm_heap');
+wait_for_stats(q{
+SELECT n_tup_del = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap'
+}, 'concurrent delete report');
+is($node->safe_psql('postgres', $table_query), "$cleared|$cleared",
+ 'DELETE counts fresh VM clearings while the reader transaction is open');
+is($reader_query->($table_query), "$pages|$pages",
+ 'cached statistics retain their earlier values');
+is($reader_query->("SELECT pg_stat_clear_snapshot(); $table_query"),
+ "$cleared|$cleared", 'clearing the statistics snapshot exposes the new counters');
+is($reader_query->('SELECT count(*) FROM vm_heap'), '1000',
+ 'refreshing statistics leaves the MVCC snapshot unchanged');
+$in = "COMMIT;\n\\q\n";
+$reader->finish;
+is($node->safe_psql('postgres', $table_query), "$cleared|$cleared",
+ 'committing the reader adds no clearings');
+is($node->safe_psql('postgres', qq{
+SELECT frozen_page_marks_cleared - $db_frozen,
+ visible_page_marks_cleared - $db_visible
+FROM pg_stat_vacuum_database WHERE datname = current_database()
+}), $cleared . '|' . ($cleared + $visible_pages),
+ 'database totals accumulate both rounds of heap clearings');
+
+my $db_totals = $node->safe_psql('postgres', $db_query);
+$node->restart;
+is($node->safe_psql('postgres', $table_query), "$cleared|$cleared",
+ 'table VM counters survive a clean restart');
+is($node->safe_psql('postgres', $visible_query), "0|$visible_pages",
+ 'counts from aborted DML survive a clean restart');
+is($node->safe_psql('postgres', $db_query), $db_totals,
+ 'database VM counters survive a clean restart');
+
+$node->safe_psql('postgres',
+ q{SELECT pg_stat_reset_single_table_counters('vm_heap'::regclass)});
+wait_for_stats(q{
+SELECT frozen_page_marks_cleared = 0 AND visible_page_marks_cleared = 0
+FROM pg_stat_vacuum_tables WHERE relname = 'vm_heap'
+}, 'relation reset');
+is($node->safe_psql('postgres', $table_query), '0|0',
+ 'relation reset clears both VM counters');
+is($node->safe_psql('postgres', $visible_query), "0|$visible_pages",
+ 'relation reset preserves another table\'s counters');
+is($node->safe_psql('postgres', $db_query), $db_totals,
+ 'relation reset preserves database totals');
+
+$node->safe_psql('postgres', 'SELECT pg_stat_reset()');
+wait_for_stats(q{
+SELECT frozen_page_marks_cleared = 0 AND visible_page_marks_cleared = 0
+FROM pg_stat_vacuum_database WHERE datname = current_database()
+}, 'database reset');
+is($node->safe_psql('postgres', $db_query), '0|0',
+ 'database reset clears both VM counters');
+is($node->safe_psql('postgres', $visible_query), '0|0',
+ 'database reset also clears relation VM counters');
+
+# The SQL interface belongs to the extension, not the system catalog.
+is($node->safe_psql('postgres', q{
+SELECT count(*) FROM pg_attribute
+WHERE attrelid IN ('pg_stat_all_tables'::regclass, 'pg_stat_database'::regclass)
+ AND attname IN ('frozen_page_marks_cleared', 'visible_page_marks_cleared')
+ AND NOT attisdropped
+}), '0', 'core statistics views retain their original columns');
+
+$node->stop;
+done_testing();
diff --git a/contrib/vacuum_stats/t/005_vacuum_interrupts.pl b/contrib/vacuum_stats/t/005_vacuum_interrupts.pl
new file mode 100644
index 00000000000..57967fc491a
--- /dev/null
+++ b/contrib/vacuum_stats/t/005_vacuum_interrupts.pl
@@ -0,0 +1,133 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+
+# Adapt v44-0004 to the PG14 collector: wait for heap processing before
+# canceling VACUUM, then wait for its deferred statistics report.
+use strict;
+use warnings FATAL => 'all';
+use PostgresNode;
+use TestLib;
+use Test::More;
+
+my $node = get_new_node('vacuum_interrupts');
+$node->init;
+$node->append_conf('postgresql.conf', "autovacuum = off\n");
+$node->start;
+$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats');
+
+my $db_count = q{SELECT vacuum_interrupt_count FROM pg_stat_vacuum_database
+ WHERE datname = current_database()};
+my $shared_count = q{SELECT vacuum_interrupt_count FROM pg_stat_vacuum_database
+ WHERE datid = 0};
+my $nrows = 1000 * ($node->safe_psql('postgres', 'SHOW block_size') / 8192);
+$node->safe_psql('postgres', qq{
+CREATE TABLE vacstat_int (id int PRIMARY KEY)
+ WITH (autovacuum_enabled = off, fillfactor = 10);
+INSERT INTO vacstat_int SELECT generate_series(1, $nrows);
+DELETE FROM vacstat_int WHERE id % 2 = 0;
+});
+
+# VERBOSE emits lower-severity messages while the vacuum error callback is
+# installed. They must not be mistaken for an interrupted vacuum.
+$node->safe_psql('postgres', 'VACUUM (VERBOSE, INDEX_CLEANUP ON) vacstat_int');
+$node->poll_query_until('postgres', q{
+SELECT vacuum_count = 1 FROM pg_stat_all_tables_internal WHERE relname = 'vacstat_int'
+}) or BAIL_OUT('successful vacuum report did not reach the collector');
+is($node->safe_psql('postgres', $db_count), '0',
+ 'successful vacuum and its VERBOSE messages do not count as errors');
+
+my ($stdout, $stderr) = ('', '');
+is($node->psql('postgres', 'SELECT 1 / 0',
+ stdout => \$stdout, stderr => \$stderr), 3,
+ 'unrelated statement raises an error');
+is($node->safe_psql('postgres', $db_count), '0',
+ 'an error outside VACUUM does not increment the counter');
+
+sub cancel_vacuum
+{
+ my ($relation, $track_counts) = @_;
+ my ($in, $out) = ('', '');
+ my $timer = IPC::Run::timeout($TestLib::timeout_default);
+ my $vac = $node->background_psql('postgres', \$in, \$out, $timer,
+ on_error_stop => 0);
+ $out = '';
+ $in = qq{
+SET application_name = 'vacuum_interrupt_test';
+SET track_counts = $track_counts;
+SET vacuum_cost_delay = '100ms';
+SET vacuum_cost_limit = 1;
+\\echo vacuum_started
+VACUUM (DISABLE_PAGE_SKIPPING) $relation;
+\\echo vacuum_done :ERROR :SQLSTATE
+};
+ pump_until($vac, $timer, \$out, qr/^vacuum_started\r?$/m)
+ or BAIL_OUT('background psql did not launch VACUUM');
+
+ # An active VACUUM query alone is insufficient: it might still be waiting
+ # for a relation lock, before the heap error callback has been installed.
+ $node->poll_query_until('postgres', qq{
+SELECT count(*) = 1
+FROM pg_stat_activity a JOIN pg_stat_progress_vacuum v USING (pid)
+WHERE a.application_name = 'vacuum_interrupt_test'
+ AND v.relid = '$relation'::regclass AND v.phase = 'scanning heap'
+ AND a.wait_event = 'VacuumDelay'
+}) or BAIL_OUT("VACUUM of $relation did not enter heap processing");
+ is($node->safe_psql('postgres', q{
+SELECT pg_cancel_backend(pid) FROM pg_stat_activity
+WHERE application_name = 'vacuum_interrupt_test'
+}), 't', "sent cancellation to VACUUM of $relation");
+ pump_until($vac, $timer, \$out, qr/^vacuum_done\b/m)
+ or BAIL_OUT('canceled VACUUM did not return');
+ like($out, qr/^vacuum_done true 57014\r?$/m,
+ "VACUUM of $relation reports query_canceled");
+ $in = "\\q\n";
+ $vac->finish;
+}
+
+for my $expected (1..2)
+{
+ cancel_vacuum('vacstat_int', 'on');
+ ok($node->poll_query_until('postgres',
+ "SELECT ($db_count) = $expected"),
+ "canceled vacuum increments database total to exactly $expected");
+}
+is($node->safe_psql('postgres', $shared_count), '0',
+ 'local relation errors leave shared-object statistics alone');
+
+cancel_vacuum('vacstat_int', 'off');
+# Drain reports, including the canceled backend's final message, before
+# asserting that a disabled counter did not change.
+$node->restart;
+is($node->safe_psql('postgres', $db_count), '2',
+ 'track_counts off suppresses counting; earlier errors survive restart');
+
+# Make pg_authid large enough to observe cost-delay waits even with 32kB
+# blocks. Shared relations report errors to datid zero, not the session's DB.
+$node->safe_psql('postgres', qq{
+DO \$\$ BEGIN
+ FOR i IN 1..$nrows LOOP
+ EXECUTE format('CREATE ROLE vacstat_role_%s', i);
+ END LOOP;
+END \$\$;
+});
+cancel_vacuum('pg_authid', 'on');
+ok($node->poll_query_until('postgres', "SELECT ($shared_count) = 1"),
+ 'shared relation error reaches the datid zero entry');
+is($node->safe_psql('postgres', $db_count), '2',
+ 'shared relation error leaves the current database total unchanged');
+
+$node->safe_psql('postgres', 'SELECT pg_stat_reset()');
+ok($node->poll_query_until('postgres', "SELECT ($db_count) = 0"),
+ 'ordinary database reset clears vacuum interruptions');
+$node->restart;
+is($node->safe_psql('postgres', $db_count), '0',
+ 'reset value survives restart');
+is($node->safe_psql('postgres', $shared_count), '1',
+ 'database reset preserves shared errors, which survive restart');
+is($node->safe_psql('postgres', q{
+SELECT count(*) FROM pg_attribute
+WHERE attrelid = 'pg_stat_vacuum_database'::regclass
+ AND attname = 'vacuum_interrupt_count' AND NOT attisdropped
+}), '1', 'extension view exposes vacuum_interrupt_count');
+
+$node->stop;
+done_testing();
diff --git a/contrib/vacuum_stats/t/006_extension_catalog.pl b/contrib/vacuum_stats/t/006_extension_catalog.pl
new file mode 100644
index 00000000000..900542c17ef
--- /dev/null
+++ b/contrib/vacuum_stats/t/006_extension_catalog.pl
@@ -0,0 +1,90 @@
+# Copyright (c) 2026, PostgreSQL Global Development Group
+
+# SQL access belongs to an extension, including counters collected while the
+# extended vacuum statistics are disabled. Check installation in another schema
+# and removal/reinstallation without altering the system statistics views.
+use strict;
+use warnings FATAL => 'all';
+use PostgresNode;
+use TestLib;
+use Test::More;
+
+my $node = get_new_node('extension_catalog');
+$node->init;
+$node->append_conf('postgresql.conf', "autovacuum = off\n");
+$node->start;
+
+my $catalog_query = q{
+SELECT c.relname, pg_get_viewdef(c.oid), a.attnum, a.attname, a.atttypid
+FROM pg_class c JOIN pg_namespace n ON n.oid = c.relnamespace
+JOIN pg_attribute a ON a.attrelid = c.oid AND a.attnum > 0
+WHERE n.nspname = 'pg_catalog' AND c.relkind = 'v'
+ AND (c.relname LIKE 'pg_stat_%' OR c.relname LIKE 'gp_stat_%')
+ORDER BY c.relname, a.attnum
+};
+my $catalog_before = $node->safe_psql('postgres', $catalog_query);
+$node->safe_psql('postgres', q{
+CREATE SCHEMA maintenance;
+CREATE EXTENSION vacuum_stats SCHEMA maintenance;
+});
+is($node->safe_psql('postgres', $catalog_query), $catalog_before,
+ 'installing the extension preserves system statistics view definitions and columns');
+is($node->safe_psql('postgres', q{
+SELECT count(*) FROM pg_proc p JOIN pg_namespace n ON n.oid = p.pronamespace
+WHERE n.nspname = 'pg_catalog'
+ AND p.proname IN ('pg_stat_get_frozen_page_marks_cleared',
+ 'pg_stat_get_visible_page_marks_cleared',
+ 'pg_stat_get_total_vacuum_time', 'pg_stat_get_total_autovacuum_time',
+ 'pg_stat_get_total_analyze_time', 'pg_stat_get_total_autoanalyze_time',
+ 'pg_stat_get_total_vacuum_delay_time', 'pg_stat_get_total_autovacuum_delay_time',
+ 'pg_stat_get_vacuum_failsafe_count',
+ 'pg_stat_get_db_frozen_page_marks_cleared',
+ 'pg_stat_get_db_visible_page_marks_cleared',
+ 'pg_stat_get_db_total_vacuum_time', 'pg_stat_get_db_total_autovacuum_time',
+ 'pg_stat_get_db_total_vacuum_delay_time', 'pg_stat_get_db_total_autovacuum_delay_time',
+ 'pg_stat_get_db_vacuum_failsafe_count', 'pg_stat_get_db_vacuum_interrupt_count')
+}), '0', 'new statistics getters are not installed as built-in functions');
+
+is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'off',
+ 'ordinary maintenance timing is tested without extended statistics storage');
+$node->safe_psql('postgres', q{
+CREATE TABLE analyze_time (id int);
+INSERT INTO analyze_time SELECT generate_series(1, 10000);
+ANALYZE analyze_time;
+});
+$node->poll_query_until('postgres', q{
+SELECT total_analyze_time > 0 FROM maintenance.pg_stat_vacuum_tables
+WHERE relname = 'analyze_time'
+}) or BAIL_OUT('ANALYZE timing report did not reach the collector');
+my $analyze_query = q{
+SELECT total_analyze_time, total_autoanalyze_time
+FROM maintenance.pg_stat_vacuum_tables WHERE relname = 'analyze_time'
+};
+my $analyze_before = $node->safe_psql('postgres', $analyze_query);
+is($node->safe_psql('postgres', q{
+SELECT total_analyze_time > 0 AND total_autoanalyze_time = 0
+FROM maintenance.gp_stat_vacuum_tables WHERE relname = 'analyze_time'
+}), 't', 'cluster view resolves extension getters in a non-default schema');
+is($node->safe_psql('postgres', q{
+SELECT count(*) FROM maintenance.pg_stat_vacuum_database
+WHERE datid = 0 AND datname IS NULL
+}), '1', 'local database view exposes the shared-relation entry');
+is($node->safe_psql('postgres', q{
+SELECT count(*) FROM maintenance.gp_stat_vacuum_database
+WHERE datid = 0 AND datname IS NULL
+}), '1', 'cluster database view exposes the shared entry once in utility mode');
+
+$node->safe_psql('postgres', 'SELECT maintenance.vacuum_stats_reset()');
+# Drain the asynchronous reset before testing fields it must preserve.
+$node->restart;
+is($node->safe_psql('postgres', $analyze_query), $analyze_before,
+ 'dedicated vacuum reset preserves ANALYZE times');
+$node->safe_psql('postgres', 'DROP EXTENSION vacuum_stats');
+is($node->safe_psql('postgres', $catalog_query), $catalog_before,
+ 'dropping the extension preserves system statistics views');
+$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats SCHEMA maintenance');
+is($node->safe_psql('postgres', $analyze_query), $analyze_before,
+ 'reinstalling the extension reads the same collector statistics');
+
+$node->stop;
+done_testing();
diff --git a/contrib/vacuum_stats/vacuum_stats--1.0.sql b/contrib/vacuum_stats/vacuum_stats--1.0.sql
new file mode 100644
index 00000000000..eea69643382
--- /dev/null
+++ b/contrib/vacuum_stats/vacuum_stats--1.0.sql
@@ -0,0 +1,588 @@
+/* contrib/vacuum_stats/vacuum_stats--1.0.sql */
+
+-- complain if script is sourced in psql, rather than via CREATE EXTENSION
+\echo Use "CREATE EXTENSION vacuum_stats" to load this file. \quit
+
+--
+-- Per-relation accessor functions (tables and indexes).
+--
+CREATE FUNCTION pg_stat_get_vacuum_tuples_deleted(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_tuples_deleted'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_vacuum_dead_tuples(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_dead_tuples'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_vacuum_pages_deleted(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_pages_deleted'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_vacuum_bytes_removed(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_bytes_removed'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_vacuum_total_file_segs(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_total_file_segs'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_vacuum_dead_pages(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_dead_pages'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_vacuum_pages_frozen(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_pages_frozen'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_vacuum_pages_all_visible(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_pages_all_visible'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_vacuum_freeze_age_count(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_freeze_age_count'
+LANGUAGE C STABLE STRICT;
+
+--
+-- Per-database accessor functions.
+--
+CREATE FUNCTION pg_stat_get_db_vacuum_tuples_deleted(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_tuples_deleted'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_db_vacuum_dead_tuples(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_dead_tuples'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_db_vacuum_pages_deleted(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_pages_deleted'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_db_vacuum_bytes_removed(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_bytes_removed'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_db_vacuum_dead_pages(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_dead_pages'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_db_vacuum_pages_frozen(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_pages_frozen'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_db_vacuum_pages_all_visible(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_pages_all_visible'
+LANGUAGE C STABLE STRICT;
+
+CREATE FUNCTION pg_stat_get_db_vacuum_freeze_age_count(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_freeze_age_count'
+LANGUAGE C STABLE STRICT;
+
+-- Access to counters stored in ordinary relation and database statistics.
+CREATE FUNCTION pg_stat_get_frozen_page_marks_cleared(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_frozen_page_marks_cleared'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_visible_page_marks_cleared(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_visible_page_marks_cleared'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_total_vacuum_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_total_vacuum_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_total_autovacuum_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_total_autovacuum_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_total_vacuum_delay_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_total_vacuum_delay_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_total_autovacuum_delay_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_total_autovacuum_delay_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_total_analyze_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_total_analyze_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_total_autoanalyze_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_total_autoanalyze_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_vacuum_failsafe_count(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_failsafe_count'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_db_frozen_page_marks_cleared(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_frozen_page_marks_cleared'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_db_visible_page_marks_cleared(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_visible_page_marks_cleared'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_db_total_vacuum_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_vacuum_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_db_total_autovacuum_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_autovacuum_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_db_total_vacuum_delay_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_vacuum_delay_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_db_total_autovacuum_delay_time(oid) RETURNS double precision
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_autovacuum_delay_time'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_db_vacuum_failsafe_count(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_failsafe_count'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+CREATE FUNCTION pg_stat_get_db_vacuum_interrupt_count(oid) RETURNS int8
+AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_interrupt_count'
+LANGUAGE C STABLE STRICT PARALLEL RESTRICTED;
+
+-- Local statistics, in milliseconds for timing columns.
+CREATE VIEW pg_stat_vacuum_tables AS
+ SELECT
+ c.oid AS relid,
+ n.nspname AS schemaname,
+ c.relname AS relname,
+ @extschema@.pg_stat_get_vacuum_tuples_deleted(c.oid) AS tuples_deleted,
+ @extschema@.pg_stat_get_vacuum_dead_tuples(c.oid) AS dead_tuples,
+ @extschema@.pg_stat_get_vacuum_pages_deleted(c.oid) AS pages_deleted,
+ @extschema@.pg_stat_get_vacuum_bytes_removed(c.oid) AS bytes_removed,
+ @extschema@.pg_stat_get_vacuum_dead_pages(c.oid) AS dead_pages,
+ @extschema@.pg_stat_get_vacuum_pages_frozen(c.oid) AS pages_frozen,
+ @extschema@.pg_stat_get_vacuum_pages_all_visible(c.oid) AS pages_all_visible,
+ @extschema@.pg_stat_get_frozen_page_marks_cleared(c.oid) AS frozen_page_marks_cleared,
+ @extschema@.pg_stat_get_visible_page_marks_cleared(c.oid) AS visible_page_marks_cleared,
+ @extschema@.pg_stat_get_vacuum_freeze_age_count(c.oid) AS freeze_age_vacuum_count,
+ @extschema@.pg_stat_get_vacuum_failsafe_count(c.oid) AS vacuum_failsafe_count,
+ @extschema@.pg_stat_get_total_vacuum_time(c.oid) AS total_vacuum_time,
+ @extschema@.pg_stat_get_total_autovacuum_time(c.oid) AS total_autovacuum_time,
+ @extschema@.pg_stat_get_total_vacuum_delay_time(c.oid) AS total_vacuum_delay_time,
+ @extschema@.pg_stat_get_total_autovacuum_delay_time(c.oid) AS total_autovacuum_delay_time,
+ @extschema@.pg_stat_get_total_analyze_time(c.oid) AS total_analyze_time,
+ @extschema@.pg_stat_get_total_autoanalyze_time(c.oid) AS total_autoanalyze_time,
+ @extschema@.pg_stat_get_vacuum_total_file_segs(c.oid) AS total_file_segs
+ FROM pg_catalog.pg_class c
+ LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
+ WHERE c.relkind IN ('r', 't', 'm');
+
+CREATE VIEW pg_stat_vacuum_indexes AS
+ SELECT
+ c.oid AS relid,
+ i.oid AS indexrelid,
+ n.nspname AS schemaname,
+ c.relname AS relname,
+ i.relname AS indexrelname,
+ @extschema@.pg_stat_get_vacuum_tuples_deleted(i.oid) AS tuples_deleted,
+ @extschema@.pg_stat_get_vacuum_dead_tuples(i.oid) AS dead_tuples,
+ @extschema@.pg_stat_get_vacuum_pages_deleted(i.oid) AS pages_deleted,
+ @extschema@.pg_stat_get_vacuum_bytes_removed(i.oid) AS bytes_removed,
+ @extschema@.pg_stat_get_vacuum_dead_pages(i.oid) AS dead_pages,
+ @extschema@.pg_stat_get_vacuum_pages_frozen(i.oid) AS pages_frozen,
+ @extschema@.pg_stat_get_vacuum_pages_all_visible(i.oid) AS pages_all_visible,
+ @extschema@.pg_stat_get_frozen_page_marks_cleared(i.oid) AS frozen_page_marks_cleared,
+ @extschema@.pg_stat_get_visible_page_marks_cleared(i.oid) AS visible_page_marks_cleared,
+ @extschema@.pg_stat_get_vacuum_freeze_age_count(i.oid) AS freeze_age_vacuum_count,
+ @extschema@.pg_stat_get_vacuum_failsafe_count(i.oid) AS vacuum_failsafe_count,
+ @extschema@.pg_stat_get_total_vacuum_time(i.oid) AS total_vacuum_time,
+ @extschema@.pg_stat_get_total_autovacuum_time(i.oid) AS total_autovacuum_time,
+ @extschema@.pg_stat_get_total_vacuum_delay_time(i.oid) AS total_vacuum_delay_time,
+ @extschema@.pg_stat_get_total_autovacuum_delay_time(i.oid) AS total_autovacuum_delay_time
+ FROM pg_catalog.pg_class c
+ JOIN pg_catalog.pg_index x ON c.oid = x.indrelid
+ JOIN pg_catalog.pg_class i ON i.oid = x.indexrelid
+ LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
+ WHERE c.relkind IN ('r', 't', 'm') AND i.relkind = 'i';
+
+CREATE VIEW pg_stat_vacuum_database AS
+ SELECT
+ d.oid AS datid,
+ d.datname AS datname,
+ @extschema@.pg_stat_get_db_vacuum_tuples_deleted(d.oid) AS tuples_deleted,
+ @extschema@.pg_stat_get_db_vacuum_dead_tuples(d.oid) AS dead_tuples,
+ @extschema@.pg_stat_get_db_vacuum_pages_deleted(d.oid) AS pages_deleted,
+ @extschema@.pg_stat_get_db_vacuum_bytes_removed(d.oid) AS bytes_removed,
+ @extschema@.pg_stat_get_db_vacuum_dead_pages(d.oid) AS dead_pages,
+ @extschema@.pg_stat_get_db_vacuum_pages_frozen(d.oid) AS pages_frozen,
+ @extschema@.pg_stat_get_db_vacuum_pages_all_visible(d.oid) AS pages_all_visible,
+ @extschema@.pg_stat_get_db_frozen_page_marks_cleared(d.oid) AS frozen_page_marks_cleared,
+ @extschema@.pg_stat_get_db_visible_page_marks_cleared(d.oid) AS visible_page_marks_cleared,
+ @extschema@.pg_stat_get_db_vacuum_freeze_age_count(d.oid) AS freeze_age_vacuum_count,
+ @extschema@.pg_stat_get_db_vacuum_failsafe_count(d.oid) AS vacuum_failsafe_count,
+ @extschema@.pg_stat_get_db_total_vacuum_time(d.oid) AS total_vacuum_time,
+ @extschema@.pg_stat_get_db_total_autovacuum_time(d.oid) AS total_autovacuum_time,
+ @extschema@.pg_stat_get_db_total_vacuum_delay_time(d.oid) AS total_vacuum_delay_time,
+ @extschema@.pg_stat_get_db_total_autovacuum_delay_time(d.oid) AS total_autovacuum_delay_time,
+ @extschema@.pg_stat_get_db_vacuum_interrupt_count(d.oid) AS vacuum_interrupt_count
+ FROM (
+ SELECT 0::oid AS oid, NULL::name AS datname
+ UNION ALL
+ SELECT oid, datname FROM pg_catalog.pg_database
+ ) d;
+
+-- Cluster views execute on the coordinator and all segments. The bodies
+-- access catalogs directly because segment functions cannot scan local views.
+-- Utility sessions return their local rows once, without a segment branch.
+CREATE FUNCTION gp_stat_get_coordinator_vacuum_tables() RETURNS SETOF RECORD AS
+$$
+ SELECT pg_catalog.gp_execution_segment() AS gp_segment_id,
+ c.oid,
+ n.nspname,
+ c.relname,
+ @extschema@.pg_stat_get_vacuum_tuples_deleted(c.oid),
+ @extschema@.pg_stat_get_vacuum_dead_tuples(c.oid),
+ @extschema@.pg_stat_get_vacuum_pages_deleted(c.oid),
+ @extschema@.pg_stat_get_vacuum_bytes_removed(c.oid),
+ @extschema@.pg_stat_get_vacuum_dead_pages(c.oid),
+ @extschema@.pg_stat_get_vacuum_pages_frozen(c.oid),
+ @extschema@.pg_stat_get_vacuum_pages_all_visible(c.oid),
+ @extschema@.pg_stat_get_frozen_page_marks_cleared(c.oid),
+ @extschema@.pg_stat_get_visible_page_marks_cleared(c.oid),
+ @extschema@.pg_stat_get_vacuum_freeze_age_count(c.oid),
+ @extschema@.pg_stat_get_vacuum_failsafe_count(c.oid),
+ @extschema@.pg_stat_get_total_vacuum_time(c.oid),
+ @extschema@.pg_stat_get_total_autovacuum_time(c.oid),
+ @extschema@.pg_stat_get_total_vacuum_delay_time(c.oid),
+ @extschema@.pg_stat_get_total_autovacuum_delay_time(c.oid),
+ @extschema@.pg_stat_get_total_analyze_time(c.oid),
+ @extschema@.pg_stat_get_total_autoanalyze_time(c.oid),
+ @extschema@.pg_stat_get_vacuum_total_file_segs(c.oid)
+ FROM pg_catalog.pg_class c
+ LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
+ WHERE c.relkind IN ('r', 't', 'm')
+$$
+LANGUAGE SQL EXECUTE ON COORDINATOR;
+
+CREATE FUNCTION gp_stat_get_segment_vacuum_tables() RETURNS SETOF RECORD AS
+$$
+ SELECT pg_catalog.gp_execution_segment() AS gp_segment_id,
+ c.oid,
+ n.nspname,
+ c.relname,
+ @extschema@.pg_stat_get_vacuum_tuples_deleted(c.oid),
+ @extschema@.pg_stat_get_vacuum_dead_tuples(c.oid),
+ @extschema@.pg_stat_get_vacuum_pages_deleted(c.oid),
+ @extschema@.pg_stat_get_vacuum_bytes_removed(c.oid),
+ @extschema@.pg_stat_get_vacuum_dead_pages(c.oid),
+ @extschema@.pg_stat_get_vacuum_pages_frozen(c.oid),
+ @extschema@.pg_stat_get_vacuum_pages_all_visible(c.oid),
+ @extschema@.pg_stat_get_frozen_page_marks_cleared(c.oid),
+ @extschema@.pg_stat_get_visible_page_marks_cleared(c.oid),
+ @extschema@.pg_stat_get_vacuum_freeze_age_count(c.oid),
+ @extschema@.pg_stat_get_vacuum_failsafe_count(c.oid),
+ @extschema@.pg_stat_get_total_vacuum_time(c.oid),
+ @extschema@.pg_stat_get_total_autovacuum_time(c.oid),
+ @extschema@.pg_stat_get_total_vacuum_delay_time(c.oid),
+ @extschema@.pg_stat_get_total_autovacuum_delay_time(c.oid),
+ @extschema@.pg_stat_get_total_analyze_time(c.oid),
+ @extschema@.pg_stat_get_total_autoanalyze_time(c.oid),
+ @extschema@.pg_stat_get_vacuum_total_file_segs(c.oid)
+ FROM pg_catalog.pg_class c
+ LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
+ WHERE c.relkind IN ('r', 't', 'm')
+ AND pg_catalog.current_setting('gp_role') <> 'utility'
+$$
+LANGUAGE SQL EXECUTE ON ALL SEGMENTS;
+
+CREATE VIEW gp_stat_vacuum_tables AS
+ SELECT * FROM gp_stat_get_coordinator_vacuum_tables() AS T
+ (gp_segment_id int,
+ relid oid,
+ schemaname name,
+ relname name,
+ tuples_deleted int8,
+ dead_tuples int8,
+ pages_deleted int8,
+ bytes_removed int8,
+ dead_pages int8,
+ pages_frozen int8,
+ pages_all_visible int8,
+ frozen_page_marks_cleared int8,
+ visible_page_marks_cleared int8,
+ freeze_age_vacuum_count int8,
+ vacuum_failsafe_count int8,
+ total_vacuum_time double precision,
+ total_autovacuum_time double precision,
+ total_vacuum_delay_time double precision,
+ total_autovacuum_delay_time double precision,
+ total_analyze_time double precision,
+ total_autoanalyze_time double precision,
+ total_file_segs int8)
+ UNION ALL
+ SELECT * FROM gp_stat_get_segment_vacuum_tables() AS T
+ (gp_segment_id int,
+ relid oid,
+ schemaname name,
+ relname name,
+ tuples_deleted int8,
+ dead_tuples int8,
+ pages_deleted int8,
+ bytes_removed int8,
+ dead_pages int8,
+ pages_frozen int8,
+ pages_all_visible int8,
+ frozen_page_marks_cleared int8,
+ visible_page_marks_cleared int8,
+ freeze_age_vacuum_count int8,
+ vacuum_failsafe_count int8,
+ total_vacuum_time double precision,
+ total_autovacuum_time double precision,
+ total_vacuum_delay_time double precision,
+ total_autovacuum_delay_time double precision,
+ total_analyze_time double precision,
+ total_autoanalyze_time double precision,
+ total_file_segs int8);
+
+CREATE FUNCTION gp_stat_get_coordinator_vacuum_indexes() RETURNS SETOF RECORD AS
+$$
+ SELECT pg_catalog.gp_execution_segment() AS gp_segment_id,
+ c.oid,
+ i.oid,
+ n.nspname,
+ c.relname,
+ i.relname,
+ @extschema@.pg_stat_get_vacuum_tuples_deleted(i.oid),
+ @extschema@.pg_stat_get_vacuum_dead_tuples(i.oid),
+ @extschema@.pg_stat_get_vacuum_pages_deleted(i.oid),
+ @extschema@.pg_stat_get_vacuum_bytes_removed(i.oid),
+ @extschema@.pg_stat_get_vacuum_dead_pages(i.oid),
+ @extschema@.pg_stat_get_vacuum_pages_frozen(i.oid),
+ @extschema@.pg_stat_get_vacuum_pages_all_visible(i.oid),
+ @extschema@.pg_stat_get_frozen_page_marks_cleared(i.oid),
+ @extschema@.pg_stat_get_visible_page_marks_cleared(i.oid),
+ @extschema@.pg_stat_get_vacuum_freeze_age_count(i.oid),
+ @extschema@.pg_stat_get_vacuum_failsafe_count(i.oid),
+ @extschema@.pg_stat_get_total_vacuum_time(i.oid),
+ @extschema@.pg_stat_get_total_autovacuum_time(i.oid),
+ @extschema@.pg_stat_get_total_vacuum_delay_time(i.oid),
+ @extschema@.pg_stat_get_total_autovacuum_delay_time(i.oid)
+ FROM pg_catalog.pg_class c
+ JOIN pg_catalog.pg_index x ON c.oid = x.indrelid
+ JOIN pg_catalog.pg_class i ON i.oid = x.indexrelid
+ LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
+ WHERE c.relkind IN ('r', 't', 'm') AND i.relkind = 'i'
+$$
+LANGUAGE SQL EXECUTE ON COORDINATOR;
+
+CREATE FUNCTION gp_stat_get_segment_vacuum_indexes() RETURNS SETOF RECORD AS
+$$
+ SELECT pg_catalog.gp_execution_segment() AS gp_segment_id,
+ c.oid,
+ i.oid,
+ n.nspname,
+ c.relname,
+ i.relname,
+ @extschema@.pg_stat_get_vacuum_tuples_deleted(i.oid),
+ @extschema@.pg_stat_get_vacuum_dead_tuples(i.oid),
+ @extschema@.pg_stat_get_vacuum_pages_deleted(i.oid),
+ @extschema@.pg_stat_get_vacuum_bytes_removed(i.oid),
+ @extschema@.pg_stat_get_vacuum_dead_pages(i.oid),
+ @extschema@.pg_stat_get_vacuum_pages_frozen(i.oid),
+ @extschema@.pg_stat_get_vacuum_pages_all_visible(i.oid),
+ @extschema@.pg_stat_get_frozen_page_marks_cleared(i.oid),
+ @extschema@.pg_stat_get_visible_page_marks_cleared(i.oid),
+ @extschema@.pg_stat_get_vacuum_freeze_age_count(i.oid),
+ @extschema@.pg_stat_get_vacuum_failsafe_count(i.oid),
+ @extschema@.pg_stat_get_total_vacuum_time(i.oid),
+ @extschema@.pg_stat_get_total_autovacuum_time(i.oid),
+ @extschema@.pg_stat_get_total_vacuum_delay_time(i.oid),
+ @extschema@.pg_stat_get_total_autovacuum_delay_time(i.oid)
+ FROM pg_catalog.pg_class c
+ JOIN pg_catalog.pg_index x ON c.oid = x.indrelid
+ JOIN pg_catalog.pg_class i ON i.oid = x.indexrelid
+ LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace
+ WHERE c.relkind IN ('r', 't', 'm') AND i.relkind = 'i'
+ AND pg_catalog.current_setting('gp_role') <> 'utility'
+$$
+LANGUAGE SQL EXECUTE ON ALL SEGMENTS;
+
+CREATE VIEW gp_stat_vacuum_indexes AS
+ SELECT * FROM gp_stat_get_coordinator_vacuum_indexes() AS T
+ (gp_segment_id int,
+ relid oid,
+ indexrelid oid,
+ schemaname name,
+ relname name,
+ indexrelname name,
+ tuples_deleted int8,
+ dead_tuples int8,
+ pages_deleted int8,
+ bytes_removed int8,
+ dead_pages int8,
+ pages_frozen int8,
+ pages_all_visible int8,
+ frozen_page_marks_cleared int8,
+ visible_page_marks_cleared int8,
+ freeze_age_vacuum_count int8,
+ vacuum_failsafe_count int8,
+ total_vacuum_time double precision,
+ total_autovacuum_time double precision,
+ total_vacuum_delay_time double precision,
+ total_autovacuum_delay_time double precision)
+ UNION ALL
+ SELECT * FROM gp_stat_get_segment_vacuum_indexes() AS T
+ (gp_segment_id int,
+ relid oid,
+ indexrelid oid,
+ schemaname name,
+ relname name,
+ indexrelname name,
+ tuples_deleted int8,
+ dead_tuples int8,
+ pages_deleted int8,
+ bytes_removed int8,
+ dead_pages int8,
+ pages_frozen int8,
+ pages_all_visible int8,
+ frozen_page_marks_cleared int8,
+ visible_page_marks_cleared int8,
+ freeze_age_vacuum_count int8,
+ vacuum_failsafe_count int8,
+ total_vacuum_time double precision,
+ total_autovacuum_time double precision,
+ total_vacuum_delay_time double precision,
+ total_autovacuum_delay_time double precision);
+
+CREATE FUNCTION gp_stat_get_coordinator_vacuum_database() RETURNS SETOF RECORD AS
+$$
+ SELECT pg_catalog.gp_execution_segment() AS gp_segment_id,
+ d.oid,
+ d.datname,
+ @extschema@.pg_stat_get_db_vacuum_tuples_deleted(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_dead_tuples(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_pages_deleted(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_bytes_removed(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_dead_pages(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_pages_frozen(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_pages_all_visible(d.oid),
+ @extschema@.pg_stat_get_db_frozen_page_marks_cleared(d.oid),
+ @extschema@.pg_stat_get_db_visible_page_marks_cleared(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_freeze_age_count(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_failsafe_count(d.oid),
+ @extschema@.pg_stat_get_db_total_vacuum_time(d.oid),
+ @extschema@.pg_stat_get_db_total_autovacuum_time(d.oid),
+ @extschema@.pg_stat_get_db_total_vacuum_delay_time(d.oid),
+ @extschema@.pg_stat_get_db_total_autovacuum_delay_time(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_interrupt_count(d.oid)
+ FROM (
+ SELECT 0::oid AS oid, NULL::name AS datname
+ UNION ALL
+ SELECT oid, datname FROM pg_catalog.pg_database
+ ) d
+$$
+LANGUAGE SQL EXECUTE ON COORDINATOR;
+
+CREATE FUNCTION gp_stat_get_segment_vacuum_database() RETURNS SETOF RECORD AS
+$$
+ SELECT pg_catalog.gp_execution_segment() AS gp_segment_id,
+ d.oid,
+ d.datname,
+ @extschema@.pg_stat_get_db_vacuum_tuples_deleted(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_dead_tuples(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_pages_deleted(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_bytes_removed(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_dead_pages(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_pages_frozen(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_pages_all_visible(d.oid),
+ @extschema@.pg_stat_get_db_frozen_page_marks_cleared(d.oid),
+ @extschema@.pg_stat_get_db_visible_page_marks_cleared(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_freeze_age_count(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_failsafe_count(d.oid),
+ @extschema@.pg_stat_get_db_total_vacuum_time(d.oid),
+ @extschema@.pg_stat_get_db_total_autovacuum_time(d.oid),
+ @extschema@.pg_stat_get_db_total_vacuum_delay_time(d.oid),
+ @extschema@.pg_stat_get_db_total_autovacuum_delay_time(d.oid),
+ @extschema@.pg_stat_get_db_vacuum_interrupt_count(d.oid)
+ FROM (
+ SELECT 0::oid AS oid, NULL::name AS datname
+ UNION ALL
+ SELECT oid, datname FROM pg_catalog.pg_database
+ ) d
+ WHERE pg_catalog.current_setting('gp_role') <> 'utility'
+$$
+LANGUAGE SQL EXECUTE ON ALL SEGMENTS;
+
+CREATE VIEW gp_stat_vacuum_database AS
+ SELECT * FROM gp_stat_get_coordinator_vacuum_database() AS T
+ (gp_segment_id int,
+ datid oid,
+ datname name,
+ tuples_deleted int8,
+ dead_tuples int8,
+ pages_deleted int8,
+ bytes_removed int8,
+ dead_pages int8,
+ pages_frozen int8,
+ pages_all_visible int8,
+ frozen_page_marks_cleared int8,
+ visible_page_marks_cleared int8,
+ freeze_age_vacuum_count int8,
+ vacuum_failsafe_count int8,
+ total_vacuum_time double precision,
+ total_autovacuum_time double precision,
+ total_vacuum_delay_time double precision,
+ total_autovacuum_delay_time double precision,
+ vacuum_interrupt_count int8)
+ UNION ALL
+ SELECT * FROM gp_stat_get_segment_vacuum_database() AS T
+ (gp_segment_id int,
+ datid oid,
+ datname name,
+ tuples_deleted int8,
+ dead_tuples int8,
+ pages_deleted int8,
+ bytes_removed int8,
+ dead_pages int8,
+ pages_frozen int8,
+ pages_all_visible int8,
+ frozen_page_marks_cleared int8,
+ visible_page_marks_cleared int8,
+ freeze_age_vacuum_count int8,
+ vacuum_failsafe_count int8,
+ total_vacuum_time double precision,
+ total_autovacuum_time double precision,
+ total_vacuum_delay_time double precision,
+ total_autovacuum_delay_time double precision,
+ vacuum_interrupt_count int8);
+
+--
+-- Resetting, for when only these counters are in the way: pg_stat_reset()
+-- and pg_stat_reset_single_table_counters() throw away the rest of the
+-- statistics of the database or the relation as well.
+--
+-- As with the rest of the statistics, resetting acts on the node it runs on,
+-- so on a cluster both the coordinator function and the segment one have to
+-- be called.
+--
+CREATE FUNCTION vacuum_stats_reset() RETURNS void
+AS 'MODULE_PATHNAME', 'vacuum_stats_reset'
+LANGUAGE C;
+
+CREATE FUNCTION vacuum_stats_reset(relid oid) RETURNS void
+AS 'MODULE_PATHNAME', 'vacuum_stats_reset_relation'
+LANGUAGE C STRICT;
+
+CREATE FUNCTION gp_vacuum_stats_reset() RETURNS SETOF void AS
+$$ SELECT @extschema@.vacuum_stats_reset() $$
+LANGUAGE SQL EXECUTE ON ALL SEGMENTS;
+
+CREATE FUNCTION gp_vacuum_stats_reset(relid oid) RETURNS SETOF void AS
+$$ SELECT @extschema@.vacuum_stats_reset($1) $$
+LANGUAGE SQL EXECUTE ON ALL SEGMENTS;
+
+REVOKE ALL ON FUNCTION vacuum_stats_reset() FROM PUBLIC;
+REVOKE ALL ON FUNCTION vacuum_stats_reset(oid) FROM PUBLIC;
+REVOKE ALL ON FUNCTION gp_vacuum_stats_reset() FROM PUBLIC;
+REVOKE ALL ON FUNCTION gp_vacuum_stats_reset(oid) FROM PUBLIC;
+
+GRANT SELECT ON pg_stat_vacuum_tables, pg_stat_vacuum_indexes,
+ pg_stat_vacuum_database, gp_stat_vacuum_tables,
+ gp_stat_vacuum_indexes, gp_stat_vacuum_database TO PUBLIC;
diff --git a/contrib/vacuum_stats/vacuum_stats.c b/contrib/vacuum_stats/vacuum_stats.c
new file mode 100644
index 00000000000..161881a4e97
--- /dev/null
+++ b/contrib/vacuum_stats/vacuum_stats.c
@@ -0,0 +1,189 @@
+/*-------------------------------------------------------------------------
+ *
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements. See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership. The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License. You may obtain a copy of the License at
+ *
+ * http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied. See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ *
+ * vacuum_stats.c
+ * Expose the vacuum counters accumulated by the statistics collector
+ * for relations (tables and indexes) and databases.
+ *
+ * The backend collects these counters in ordinary statistics entries and,
+ * when enabled, extended vacuum statistics. SQL access belongs to this
+ * extension; read both groups through the regular pgstat fetch API.
+ *
+ * contrib/vacuum_stats/vacuum_stats.c
+ *
+ *-------------------------------------------------------------------------
+ */
+#include "postgres.h"
+
+#include "fmgr.h"
+#include "pgstat.h"
+
+PG_MODULE_MAGIC;
+
+/* Ordinary counters follow track_counts, independently of extended tracking. */
+#define DEFINE_REL_COUNTER_FUNC(funcname, field) \
+PG_FUNCTION_INFO_V1(funcname); \
+Datum \
+funcname(PG_FUNCTION_ARGS) \
+{ \
+ PgStat_StatTabEntry *entry = pgstat_fetch_stat_tabentry(PG_GETARG_OID(0)); \
+ PG_RETURN_INT64(entry ? (int64) entry->field : 0); \
+}
+
+#define DEFINE_DB_COUNTER_FUNC(funcname, field) \
+PG_FUNCTION_INFO_V1(funcname); \
+Datum \
+funcname(PG_FUNCTION_ARGS) \
+{ \
+ PgStat_StatDBEntry *entry = pgstat_fetch_stat_dbentry(PG_GETARG_OID(0)); \
+ PG_RETURN_INT64(entry ? (int64) entry->field : 0); \
+}
+
+/* Store microseconds in the collector and expose milliseconds without rounding. */
+#define DEFINE_REL_TIME_FUNC(funcname, field) \
+PG_FUNCTION_INFO_V1(funcname); \
+Datum \
+funcname(PG_FUNCTION_ARGS) \
+{ \
+ PgStat_StatTabEntry *entry = pgstat_fetch_stat_tabentry(PG_GETARG_OID(0)); \
+ PG_RETURN_FLOAT8(entry ? (double) entry->field / 1000.0 : 0); \
+}
+
+#define DEFINE_DB_TIME_FUNC(funcname, field) \
+PG_FUNCTION_INFO_V1(funcname); \
+Datum \
+funcname(PG_FUNCTION_ARGS) \
+{ \
+ PgStat_StatDBEntry *entry = pgstat_fetch_stat_dbentry(PG_GETARG_OID(0)); \
+ PG_RETURN_FLOAT8(entry ? (double) entry->field / 1000.0 : 0); \
+}
+
+DEFINE_REL_COUNTER_FUNC(pg_stat_get_frozen_page_marks_cleared, frozen_page_marks_cleared)
+DEFINE_REL_COUNTER_FUNC(pg_stat_get_visible_page_marks_cleared, visible_page_marks_cleared)
+DEFINE_REL_TIME_FUNC(pg_stat_get_total_vacuum_time, total_vacuum_time)
+DEFINE_REL_TIME_FUNC(pg_stat_get_total_autovacuum_time, total_autovacuum_time)
+DEFINE_REL_TIME_FUNC(pg_stat_get_total_vacuum_delay_time, total_vacuum_delay_time)
+DEFINE_REL_TIME_FUNC(pg_stat_get_total_autovacuum_delay_time, total_autovacuum_delay_time)
+DEFINE_REL_TIME_FUNC(pg_stat_get_total_analyze_time, total_analyze_time)
+DEFINE_REL_TIME_FUNC(pg_stat_get_total_autoanalyze_time, total_autoanalyze_time)
+DEFINE_REL_COUNTER_FUNC(pg_stat_get_vacuum_failsafe_count, vacuum_failsafe_count)
+DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_frozen_page_marks_cleared, n_frozen_page_marks_cleared)
+DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_visible_page_marks_cleared, n_visible_page_marks_cleared)
+DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_vacuum_time, total_vacuum_time)
+DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_autovacuum_time, total_autovacuum_time)
+DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_vacuum_delay_time, total_vacuum_delay_time)
+DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_autovacuum_delay_time, total_autovacuum_delay_time)
+DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_vacuum_failsafe_count, vacuum_failsafe_count)
+DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_vacuum_interrupt_count, vacuum_interrupt_count)
+
+/* Fetch the relation's vacuum counters, or NULL when unavailable. */
+static PgStat_VacuumStats *
+fetch_rel_vacuum_stats(Oid relid)
+{
+ return pgstat_fetch_stat_vacuum_stats(relid);
+}
+
+/*
+ * Fetch the per-database vacuum counters, or NULL if the statistics
+ * collector has no entry for the database.
+ */
+static PgStat_VacuumStats *
+fetch_db_vacuum_stats(Oid dbid)
+{
+ PgStat_StatDBEntry *dbentry;
+
+ if (!pgstat_track_vacuum_statistics)
+ return NULL;
+
+ dbentry = pgstat_fetch_stat_dbentry(dbid);
+ if (dbentry == NULL)
+ return NULL;
+
+ return &dbentry->n_vacuum_stats;
+}
+
+#define DEFINE_REL_VACSTAT_FUNC(funcname, field) \
+PG_FUNCTION_INFO_V1(funcname); \
+Datum \
+funcname(PG_FUNCTION_ARGS) \
+{ \
+ Oid relid = PG_GETARG_OID(0); \
+ PgStat_VacuumStats *stats = fetch_rel_vacuum_stats(relid); \
+\
+ PG_RETURN_INT64(stats ? (int64) stats->field : 0); \
+}
+
+#define DEFINE_DB_VACSTAT_FUNC(funcname, field) \
+PG_FUNCTION_INFO_V1(funcname); \
+Datum \
+funcname(PG_FUNCTION_ARGS) \
+{ \
+ Oid dbid = PG_GETARG_OID(0); \
+ PgStat_VacuumStats *stats = fetch_db_vacuum_stats(dbid); \
+\
+ PG_RETURN_INT64(stats ? (int64) stats->field : 0); \
+}
+
+DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_tuples_deleted, tuples_deleted)
+DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_dead_tuples, dead_tuples)
+DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_pages_deleted, pages_deleted)
+DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_bytes_removed, bytes_removed)
+DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_total_file_segs, total_file_segs)
+DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_dead_pages, dead_pages)
+DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_pages_frozen, pages_frozen)
+DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_pages_all_visible, pages_all_visible)
+DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_freeze_age_count, freeze_age_vacuum_count)
+
+DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_tuples_deleted, tuples_deleted)
+DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_dead_tuples, dead_tuples)
+DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_deleted, pages_deleted)
+DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_bytes_removed, bytes_removed)
+DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_dead_pages, dead_pages)
+DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_frozen, pages_frozen)
+DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_all_visible, pages_all_visible)
+DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_freeze_age_count, freeze_age_vacuum_count)
+
+/*
+ * Throw away the vacuum counters of one relation, or of the whole database,
+ * without touching the rest of the statistics -- which is what
+ * pg_stat_reset() and pg_stat_reset_single_table_counters() would do.
+ *
+ * Like the other resetting functions this acts on the node it runs on, so on
+ * a cluster it has to be dispatched to the segments as well; see
+ * gp_vacuum_stats_reset() in the extension script.
+ */
+PG_FUNCTION_INFO_V1(vacuum_stats_reset);
+Datum
+vacuum_stats_reset(PG_FUNCTION_ARGS)
+{
+ pgstat_reset_vacuum_stats(InvalidOid, true);
+
+ PG_RETURN_VOID();
+}
+
+PG_FUNCTION_INFO_V1(vacuum_stats_reset_relation);
+Datum
+vacuum_stats_reset_relation(PG_FUNCTION_ARGS)
+{
+ Oid relid = PG_GETARG_OID(0);
+
+ pgstat_reset_vacuum_stats(relid, false);
+
+ PG_RETURN_VOID();
+}
diff --git a/contrib/vacuum_stats/vacuum_stats.control b/contrib/vacuum_stats/vacuum_stats.control
new file mode 100644
index 00000000000..5f80449984d
--- /dev/null
+++ b/contrib/vacuum_stats/vacuum_stats.control
@@ -0,0 +1,5 @@
+# vacuum_stats extension
+comment = 'per-relation and per-database vacuum statistics'
+default_version = '1.0'
+module_pathname = '$libdir/vacuum_stats'
+relocatable = false
diff --git a/doc/src/sgml/config.sgml b/doc/src/sgml/config.sgml
index 0396704f3d8..ef4e256834b 100644
--- a/doc/src/sgml/config.sgml
+++ b/doc/src/sgml/config.sgml
@@ -7601,6 +7601,26 @@ COPY postgres_log FROM '/full/path/to/logfile.csv' WITH csv;
+
+ track_cost_delay_timing (boolean)
+
+ track_cost_delay_timing configuration parameter
+
+
+
+
+ Enables timing of cost-based vacuum delay (see
+ ). This parameter
+ is off by default, as it will repeatedly query the operating system for
+ the current time, which may cause significant overhead on some
+ platforms. You can use the tool to
+ measure the overhead of timing on your system. The measured time is
+ included in cumulative vacuum statistics. Only superusers can change
+ this setting.
+
+
+
+
track_io_timing (boolean)
diff --git a/src/backend/access/aocs/aocs_compaction.c b/src/backend/access/aocs/aocs_compaction.c
index e6ecea73de1..41919408c01 100644
--- a/src/backend/access/aocs/aocs_compaction.c
+++ b/src/backend/access/aocs/aocs_compaction.c
@@ -323,7 +323,7 @@ AOCSSegmentFileFullCompaction(Relation aorel,
tupleCount++;
if (VacuumCostActive && tupleCount % tuplePerPage == 0)
{
- vacuum_delay_point();
+ vacuum_delay_point(false);
}
/*
diff --git a/src/backend/access/aocs/aocsam_handler.c b/src/backend/access/aocs/aocsam_handler.c
index 4b3cd2a52ef..83bba9e0443 100644
--- a/src/backend/access/aocs/aocsam_handler.c
+++ b/src/backend/access/aocs/aocsam_handler.c
@@ -1681,7 +1681,7 @@ aoco_acquire_sample_rows(Relation onerel, int elevel, HeapTuple *rows,
{
aocoscan->targrow = RowSampler_Next(&rs);
- vacuum_delay_point();
+ vacuum_delay_point(true);
if (aocs_get_target_tuple(aocoscan, aocoscan->targrow, slot))
{
diff --git a/src/backend/access/appendonly/aomd.c b/src/backend/access/appendonly/aomd.c
index 342e7771b7b..36dfa5d71f5 100644
--- a/src/backend/access/appendonly/aomd.c
+++ b/src/backend/access/appendonly/aomd.c
@@ -221,7 +221,8 @@ TruncateAOSegmentFile(File fd, Relation rel, int32 segFileNum, int64 offset, AOV
/* report heap-equivalent blocks vacuumed */
vacrelstats->nbytes_truncated += filesize_before - offset;
pgstat_progress_update_param(PROGRESS_VACUUM_HEAP_BLKS_VACUUMED,
- RelationGuessNumberOfBlocksFromSize(vacrelstats->nbytes_truncated));
+ vacrelstats->nbytes_truncated / BLCKSZ +
+ (vacrelstats->nbytes_truncated % BLCKSZ != 0));
}
if (XLogIsNeeded() && RelationNeedsWAL(rel))
diff --git a/src/backend/access/appendonly/appendonly_compaction.c b/src/backend/access/appendonly/appendonly_compaction.c
index 05b2143b246..5ad525883b4 100644
--- a/src/backend/access/appendonly/appendonly_compaction.c
+++ b/src/backend/access/appendonly/appendonly_compaction.c
@@ -506,7 +506,7 @@ AppendOnlySegmentFileFullCompaction(Relation aorel,
tupleCount++;
if (VacuumCostActive && tupleCount % tuplePerPage == 0)
{
- vacuum_delay_point();
+ vacuum_delay_point(false);
}
}
diff --git a/src/backend/access/appendonly/appendonlyam_handler.c b/src/backend/access/appendonly/appendonlyam_handler.c
index 715cebd7579..ad299fecb51 100644
--- a/src/backend/access/appendonly/appendonlyam_handler.c
+++ b/src/backend/access/appendonly/appendonlyam_handler.c
@@ -1563,7 +1563,7 @@ appendonly_acquire_sample_rows(Relation onerel, int elevel, HeapTuple *rows,
{
aoscan->targrow = RowSampler_Next(&rs);
- vacuum_delay_point();
+ vacuum_delay_point(true);
if (appendonly_get_target_tuple(aoscan, aoscan->targrow, slot))
{
diff --git a/src/backend/access/gin/ginfast.c b/src/backend/access/gin/ginfast.c
index 3f84e90b260..317041a454f 100644
--- a/src/backend/access/gin/ginfast.c
+++ b/src/backend/access/gin/ginfast.c
@@ -894,7 +894,7 @@ ginInsertCleanup(GinState *ginstate, bool full_clean,
*/
processPendingPage(&accum, &datums, page, FirstOffsetNumber);
- vacuum_delay_point();
+ vacuum_delay_point(false);
/*
* Is it time to flush memory to disk? Flush if we are at the end of
@@ -931,7 +931,7 @@ ginInsertCleanup(GinState *ginstate, bool full_clean,
{
ginEntryInsert(ginstate, attnum, key, category,
list, nlist, NULL);
- vacuum_delay_point();
+ vacuum_delay_point(false);
}
/*
@@ -1004,7 +1004,7 @@ ginInsertCleanup(GinState *ginstate, bool full_clean,
/*
* Read next page in pending list
*/
- vacuum_delay_point();
+ vacuum_delay_point(false);
buffer = ReadBuffer(index, blkno);
LockBuffer(buffer, GIN_SHARE);
page = BufferGetPage(buffer);
diff --git a/src/backend/access/gin/ginvacuum.c b/src/backend/access/gin/ginvacuum.c
index a276eb020b5..9d0d5218392 100644
--- a/src/backend/access/gin/ginvacuum.c
+++ b/src/backend/access/gin/ginvacuum.c
@@ -663,12 +663,12 @@ ginbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats,
UnlockReleaseBuffer(buffer);
}
- vacuum_delay_point();
+ vacuum_delay_point(false);
for (i = 0; i < nRoot; i++)
{
ginVacuumPostingTree(&gvs, rootOfPostingTree[i]);
- vacuum_delay_point();
+ vacuum_delay_point(false);
}
if (blkno == InvalidBlockNumber) /* rightmost page */
@@ -749,7 +749,7 @@ ginvacuumcleanup(IndexVacuumInfo *info, IndexBulkDeleteResult *stats)
Buffer buffer;
Page page;
- vacuum_delay_point();
+ vacuum_delay_point(false);
buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno,
RBM_NORMAL, info->strategy);
diff --git a/src/backend/access/gist/gistvacuum.c b/src/backend/access/gist/gistvacuum.c
index 0663193531a..8a9805207a8 100644
--- a/src/backend/access/gist/gistvacuum.c
+++ b/src/backend/access/gist/gistvacuum.c
@@ -277,7 +277,7 @@ gistvacuumpage(GistVacState *vstate, BlockNumber blkno, BlockNumber orig_blkno)
recurse_to = InvalidBlockNumber;
/* call vacuum_delay_point while not holding any buffer lock */
- vacuum_delay_point();
+ vacuum_delay_point(false);
buffer = ReadBufferExtended(rel, MAIN_FORKNUM, blkno, RBM_NORMAL,
info->strategy);
diff --git a/src/backend/access/hash/hash.c b/src/backend/access/hash/hash.c
index 9ab8b829a1e..470daa022dc 100644
--- a/src/backend/access/hash/hash.c
+++ b/src/backend/access/hash/hash.c
@@ -730,7 +730,7 @@ hashbucketcleanup(Relation rel, Bucket cur_bucket, Buffer bucket_buf,
bool retain_pin = false;
bool clear_dead_marking = false;
- vacuum_delay_point();
+ vacuum_delay_point(false);
page = BufferGetPage(buf);
opaque = (HashPageOpaque) PageGetSpecialPointer(page);
diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c
index 44538b626c6..97c03036656 100644
--- a/src/backend/access/heap/vacuumlazy.c
+++ b/src/backend/access/heap/vacuumlazy.c
@@ -370,6 +370,10 @@ typedef struct LVRelState
BlockNumber pages_removed; /* pages remove by truncation */
BlockNumber lpdead_item_pages; /* # pages with LP_DEAD items */
BlockNumber nonempty_pages; /* actually, last nonempty page + 1 */
+ /* Counters reported as the relation's vacuum statistics */
+ BlockNumber dead_pages; /* pages left with unremovable dead tuples */
+ BlockNumber pages_frozen; /* pages where we froze tuples */
+ BlockNumber pages_all_visible; /* pages we marked all-visible */
/* Statistics output by us, for table */
double new_rel_tuples; /* new estimated total # of tuples */
@@ -512,6 +516,10 @@ heap_vacuum_rel(Relation rel, VacuumParams *params,
write_rate;
bool aggressive; /* should we scan all unfrozen pages? */
bool scanned_all_unfrozen; /* actually scanned all such pages? */
+ bool freeze_age_vacuum; /* aggressive due to freeze age? */
+ instr_time vacstart;
+ int64 startdelaytime;
+ instr_time vacend;
char **indnames = NULL;
TransactionId xidFullScanLimit;
MultiXactId mxactFullScanLimit;
@@ -527,11 +535,17 @@ heap_vacuum_rel(Relation rel, VacuumParams *params,
TransactionId FreezeLimit;
MultiXactId MultiXactCutoff;
+ /* Used for instrumentation and cumulative maintenance statistics. */
+ starttime = GetCurrentTimestamp();
+
+ /* measure elapsed and delay time for the vacuum statistics */
+ INSTR_TIME_SET_CURRENT(vacstart);
+ startdelaytime = VacuumDelayTime;
+
/* measure elapsed time iff autovacuum logging requires it */
if (IsAutoVacuumWorkerProcess() && params->log_min_duration >= 0)
{
pg_rusage_init(&ru0);
- starttime = GetCurrentTimestamp();
if (track_io_timing)
{
startreadtime = pgStatBlockReadTime;
@@ -574,6 +588,14 @@ heap_vacuum_rel(Relation rel, VacuumParams *params,
xidFullScanLimit);
aggressive |= MultiXactIdPrecedesOrEquals(rel->rd_rel->relminmxid,
mxactFullScanLimit);
+
+ /*
+ * Remember whether the freeze table age made this run aggressive.
+ * DISABLE_PAGE_SKIPPING can also force an aggressive scan, but does
+ * not by itself contribute to freeze_age_vacuum_count.
+ */
+ freeze_age_vacuum = aggressive;
+
if (params->options & VACOPT_DISABLE_PAGE_SKIPPING)
aggressive = true;
@@ -752,7 +774,42 @@ heap_vacuum_rel(Relation rel, VacuumParams *params,
pgstat_report_vacuum(RelationGetRelid(rel),
rel->rd_rel->relisshared,
Max(new_live_tuples, 0),
- vacrel->new_dead_tuples);
+ vacrel->new_dead_tuples, starttime,
+ VacuumDelayTime - startdelaytime, vacrel->failsafe_active);
+
+ /* report the per-vacuum counters as well */
+ {
+ PgStat_VacuumStats vacstats;
+ PgStat_Counter elapsedtime;
+
+ MemSet(&vacstats, 0, sizeof(vacstats));
+ vacstats.tuples_deleted = (PgStat_Counter) vacrel->tuples_deleted;
+ vacstats.dead_tuples = (PgStat_Counter) vacrel->new_dead_tuples;
+ vacstats.pages_deleted = (PgStat_Counter) vacrel->pages_removed;
+ vacstats.bytes_removed = (PgStat_Counter) vacrel->pages_removed * BLCKSZ;
+ vacstats.dead_pages = (PgStat_Counter) vacrel->dead_pages;
+ vacstats.pages_frozen = (PgStat_Counter) vacrel->pages_frozen;
+ vacstats.pages_all_visible = (PgStat_Counter) vacrel->pages_all_visible;
+ vacstats.freeze_age_vacuum_count = freeze_age_vacuum ? 1 : 0;
+
+ INSTR_TIME_SET_CURRENT(vacend);
+ INSTR_TIME_SUBTRACT(vacend, vacstart);
+ elapsedtime = (PgStat_Counter) INSTR_TIME_GET_MICROSEC(vacend);
+
+ ereport(elevel,
+ (errmsg("table \"%s\": vacuum statistics", vacrel->relname),
+ errdetail("elapsed: %.3f ms, cost-based delay: %.3f ms\n"
+ "aggressive scan required by freeze age: %s",
+ elapsedtime / 1000.0,
+ (VacuumDelayTime - startdelaytime) / 1000.0,
+ freeze_age_vacuum ? _("yes") : _("no"))));
+
+ pgstat_report_vacstats(RelationGetRelid(rel),
+ rel->rd_rel->relisshared,
+ false,
+ &vacstats);
+ }
+
pgstat_progress_end_command();
/* and log the action if appropriate */
@@ -818,6 +875,14 @@ heap_vacuum_rel(Relation rel, VacuumParams *params,
vacrel->rel_pages,
vacrel->pinskipped_pages,
vacrel->frozenskipped_pages);
+ appendStringInfo(&buf, _("pages with dead tuples not yet removable: %u\n"),
+ vacrel->dead_pages);
+ appendStringInfo(&buf, _("pages with tuples frozen: %u\n"),
+ vacrel->pages_frozen);
+ appendStringInfo(&buf, _("pages marked all-visible: %u\n"),
+ vacrel->pages_all_visible);
+ appendStringInfo(&buf, _("aggressive scan required by freeze age: %s\n"),
+ freeze_age_vacuum ? _("yes") : _("no"));
appendStringInfo(&buf,
_("tuples: %lld removed, %lld remain, %lld are dead but not yet removable, oldest xmin: %u\n"),
(long long) vacrel->tuples_deleted,
@@ -858,8 +923,9 @@ heap_vacuum_rel(Relation rel, VacuumParams *params,
continue;
appendStringInfo(&buf,
- _("index \"%s\": pages: %u in total, %u newly deleted, %u currently deleted, %u reusable\n"),
+ _("index \"%s\": tuples: %.0f removed; pages: %u in total, %u newly deleted, %u currently deleted, %u reusable\n"),
indnames[i],
+ istat->tuples_removed,
istat->num_pages,
istat->pages_newly_deleted,
istat->pages_deleted,
@@ -885,6 +951,8 @@ heap_vacuum_rel(Relation rel, VacuumParams *params,
(long long) walusage.wal_records,
(long long) walusage.wal_fpi,
(unsigned long long) walusage.wal_bytes);
+ appendStringInfo(&buf, _("cost-based delay: %.3f ms\n"),
+ (VacuumDelayTime - startdelaytime) / 1000.0);
appendStringInfo(&buf, _("system usage: %s"), pg_rusage_show(&ru0));
ereport(LOG,
@@ -1075,7 +1143,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive)
if ((vmstatus & VISIBILITYMAP_ALL_VISIBLE) == 0)
break;
}
- vacuum_delay_point();
+ vacuum_delay_point(false);
next_unskippable_block++;
}
}
@@ -1127,7 +1195,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive)
if ((vmskipflags & VISIBILITYMAP_ALL_VISIBLE) == 0)
break;
}
- vacuum_delay_point();
+ vacuum_delay_point(false);
next_unskippable_block++;
}
}
@@ -1177,7 +1245,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive)
all_visible_according_to_vm = true;
}
- vacuum_delay_point();
+ vacuum_delay_point(false);
/*
* Regularly check if wraparound failsafe should trigger.
@@ -1391,6 +1459,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive)
visibilitymap_set(vacrel->rel, blkno, buf, InvalidXLogRecPtr,
vmbuffer, InvalidTransactionId,
VISIBILITYMAP_ALL_VISIBLE | VISIBILITYMAP_ALL_FROZEN);
+ vacrel->pages_all_visible++;
END_CRIT_SECTION();
}
@@ -1502,6 +1571,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive)
visibilitymap_set(vacrel->rel, blkno, buf, InvalidXLogRecPtr,
vmbuffer, prunestate.visibility_cutoff_xid,
flags);
+ vacrel->pages_all_visible++;
}
/*
@@ -1685,6 +1755,12 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive)
appendStringInfo(&buf,
_("%lld dead row versions cannot be removed yet, oldest xmin: %u\n"),
(long long) vacrel->new_dead_tuples, vacrel->OldestXmin);
+ appendStringInfo(&buf, _("pages with dead tuples not yet removable: %u\n"),
+ vacrel->dead_pages);
+ appendStringInfo(&buf, _("pages with tuples frozen: %u\n"),
+ vacrel->pages_frozen);
+ appendStringInfo(&buf, _("pages marked all-visible: %u\n"),
+ vacrel->pages_all_visible);
appendStringInfo(&buf, ngettext("Skipped %u page due to buffer pins, ",
"Skipped %u pages due to buffer pins, ",
vacrel->pinskipped_pages),
@@ -1992,6 +2068,8 @@ lazy_scan_prune(LVRelState *vacrel,
{
Assert(prunestate->hastup);
+ vacrel->pages_frozen++;
+
/*
* At least one tuple with storage needs to be frozen -- execute that
* now.
@@ -2092,6 +2170,10 @@ lazy_scan_prune(LVRelState *vacrel,
dead_tuples->num_tuples);
}
+ /* Remember pages that keep dead tuples we could not remove yet */
+ if (new_dead_tuples > 0)
+ vacrel->dead_pages++;
+
/* Finally, add page-local counts to whole-VACUUM counts */
vacrel->tuples_deleted += tuples_deleted;
vacrel->lpdead_items += lpdead_items;
@@ -2374,7 +2456,7 @@ lazy_vacuum_heap_rel(LVRelState *vacrel)
Page page;
Size freespace;
- vacuum_delay_point();
+ vacuum_delay_point(false);
tblk = ItemPointerGetBlockNumber(&vacrel->dead_tuples->itemptrs[tupindex]);
vacrel->blkno = tblk;
@@ -2544,8 +2626,12 @@ lazy_vacuum_heap_page(LVRelState *vacrel, BlockNumber blkno, Buffer buffer,
Assert(BufferIsValid(*vmbuffer));
if (flags != 0)
+ {
visibilitymap_set(vacrel->rel, blkno, buffer, InvalidXLogRecPtr,
*vmbuffer, visibility_cutoff_xid, flags);
+ if (flags & VISIBILITYMAP_ALL_VISIBLE)
+ vacrel->pages_all_visible++;
+ }
}
/* Revert to the previous phase information for error traceback */
@@ -3044,6 +3130,81 @@ lazy_cleanup_all_indexes(LVRelState *vacrel)
}
}
+/*
+ * lazy_index_vacstats_start() -- remember where an index vacuum call starts,
+ * for lazy_index_vacstats_finish().
+ *
+ * The counters in istat accumulate over all the calls made for the index
+ * during one vacuum, so we report what each call added to them. That keeps
+ * the work of the bulk deletion passes accounted for even when the cleanup
+ * is skipped, and keeps it from being counted twice when it is not.
+ */
+static IndexBulkDeleteResult
+lazy_index_vacstats_start(IndexBulkDeleteResult *istat,
+ instr_time *starttime, int64 *startdelaytime)
+{
+ IndexBulkDeleteResult before;
+
+ if (istat)
+ before = *istat;
+ else
+ MemSet(&before, 0, sizeof(before));
+
+ INSTR_TIME_SET_CURRENT(*starttime);
+ *startdelaytime = VacuumDelayTime;
+
+ return before;
+}
+
+/*
+ * lazy_index_vacstats_finish() -- finish measuring one index vacuum call.
+ *
+ * pages_deleted and pages_free describe the whole index as the call left it,
+ * not what it did, so they are only looked at after the cleanup, which comes
+ * last: the deleted pages that are not reusable yet are the index's dead
+ * pages.
+ */
+static void
+lazy_index_vacstats_finish(Relation indrel, IndexBulkDeleteResult *istat,
+ IndexBulkDeleteResult *before, bool cleanup,
+ instr_time starttime, int64 startdelaytime)
+{
+ PgStat_VacuumStats vacstats;
+ instr_time endtime;
+
+ INSTR_TIME_SET_CURRENT(endtime);
+ INSTR_TIME_SUBTRACT(endtime, starttime);
+
+ MemSet(&vacstats, 0, sizeof(vacstats));
+ if (istat)
+ {
+ vacstats.tuples_deleted =
+ (PgStat_Counter) (istat->tuples_removed - before->tuples_removed);
+ /*
+ * Access methods accumulate pages_newly_deleted over the calls,
+ * so report only the work performed by this call.
+ */
+ if (istat->pages_newly_deleted >= before->pages_newly_deleted)
+ vacstats.pages_deleted = (PgStat_Counter)
+ (istat->pages_newly_deleted - before->pages_newly_deleted);
+ else
+ vacstats.pages_deleted = (PgStat_Counter) istat->pages_newly_deleted;
+ if (cleanup && istat->pages_deleted > istat->pages_free)
+ vacstats.dead_pages =
+ (PgStat_Counter) (istat->pages_deleted - istat->pages_free);
+ }
+
+ pgstat_report_index_vacuum_time(indrel,
+ (PgStat_Counter) INSTR_TIME_GET_MICROSEC(endtime),
+ VacuumDelayTime - startdelaytime,
+ IsAutoVacuumWorkerProcess());
+
+ pgstat_report_vacstats(RelationGetRelid(indrel),
+ indrel->rd_rel->relisshared,
+ true,
+ &vacstats);
+}
+
/*
* lazy_vacuum_one_index() -- vacuum index relation.
*
@@ -3062,6 +3223,9 @@ lazy_vacuum_one_index(Relation indrel, IndexBulkDeleteResult *istat,
IndexVacuumInfo ivinfo;
PGRUsage ru0;
LVSavedErrInfo saved_err_info;
+ IndexBulkDeleteResult istat_before;
+ instr_time starttime;
+ int64 startdelaytime;
pg_rusage_init(&ru0);
@@ -3086,13 +3250,19 @@ lazy_vacuum_one_index(Relation indrel, IndexBulkDeleteResult *istat,
InvalidBlockNumber, InvalidOffsetNumber);
/* Do bulk deletion */
+ istat_before = lazy_index_vacstats_start(istat, &starttime,
+ &startdelaytime);
istat = index_bulk_delete(&ivinfo, istat, lazy_tid_reaped,
(void *) vacrel->dead_tuples);
+ lazy_index_vacstats_finish(indrel, istat, &istat_before, false,
+ starttime, startdelaytime);
ereport(elevel,
(errmsg("scanned index \"%s\" to remove %d row versions",
vacrel->indname, vacrel->dead_tuples->num_tuples),
- errdetail_internal("%s", pg_rusage_show(&ru0))));
+ errdetail("cost-based delay: %.3f ms\n%s",
+ (VacuumDelayTime - startdelaytime) / 1000.0,
+ pg_rusage_show(&ru0))));
/* Revert to the previous phase information for error traceback */
restore_vacuum_error_info(vacrel, &saved_err_info);
@@ -3118,6 +3288,9 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat,
IndexVacuumInfo ivinfo;
PGRUsage ru0;
LVSavedErrInfo saved_err_info;
+ IndexBulkDeleteResult istat_before;
+ instr_time starttime;
+ int64 startdelaytime;
pg_rusage_init(&ru0);
@@ -3142,7 +3315,11 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat,
VACUUM_ERRCB_PHASE_INDEX_CLEANUP,
InvalidBlockNumber, InvalidOffsetNumber);
+ istat_before = lazy_index_vacstats_start(istat, &starttime,
+ &startdelaytime);
istat = index_vacuum_cleanup(&ivinfo, istat);
+ lazy_index_vacstats_finish(indrel, istat, &istat_before, true,
+ starttime, startdelaytime);
if (istat)
{
@@ -3154,10 +3331,12 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat,
errdetail("%.0f index row versions were removed.\n"
"%u index pages were newly deleted.\n"
"%u index pages are currently deleted, of which %u are currently reusable.\n"
+ "cost-based delay: %.3f ms\n"
"%s.",
(istat)->tuples_removed,
(istat)->pages_newly_deleted,
(istat)->pages_deleted, (istat)->pages_free,
+ (VacuumDelayTime - startdelaytime) / 1000.0,
pg_rusage_show(&ru0))));
}
@@ -4307,6 +4486,17 @@ vacuum_error_callback(void *arg)
{
LVRelState *errinfo = arg;
+ /*
+ * If an actual ERROR (not a lower-severity report that merely carries
+ * this vacuum error context) is being raised while we have a relation in
+ * hand, record at the database level that a vacuum was interrupted. Any
+ * error here aborts the vacuum, so the exact phase does not matter. We
+ * are inside the error handler, so this only bumps a counter; the
+ * statistics are updated at the next pgstat_report_stat().
+ */
+ if (errinfo->rel != NULL && geterrlevel() == ERROR)
+ pgstat_count_vacuum_error(errinfo->rel->rd_rel->relisshared);
+
switch (errinfo->phase)
{
case VACUUM_ERRCB_PHASE_SCAN_HEAP:
diff --git a/src/backend/access/heap/visibilitymap.c b/src/backend/access/heap/visibilitymap.c
index 7d252fa94ab..4663ade3aa3 100644
--- a/src/backend/access/heap/visibilitymap.c
+++ b/src/backend/access/heap/visibilitymap.c
@@ -90,6 +90,7 @@
#include "access/visibilitymap.h"
#include "access/xlog.h"
#include "miscadmin.h"
+#include "pgstat.h"
#include "port/pg_bitutils.h"
#include "storage/bufmgr.h"
#include "storage/lmgr.h"
@@ -159,10 +160,24 @@ visibilitymap_clear(Relation rel, BlockNumber heapBlk, Buffer buf, uint8 flags)
if (map[mapByte] & mask)
{
+ uint8 cleared_bits = (map[mapByte] & mask) >> mapOffset;
+
map[mapByte] &= ~mask;
MarkBufferDirty(buf);
cleared = true;
+
+ /*
+ * Count the pages that just lost their all-visible/all-frozen status
+ * for pg_stat_all_tables and pg_stat_database.
+ * The counters are delivered with the regular relation statistics, so
+ * nothing is counted during recovery, where rel is a fake relcache
+ * entry without a pgstat entry.
+ */
+ if (cleared_bits & VISIBILITYMAP_ALL_VISIBLE)
+ pgstat_count_visible_page_marks_cleared(rel);
+ if (cleared_bits & VISIBILITYMAP_ALL_FROZEN)
+ pgstat_count_frozen_page_marks_cleared(rel);
}
LockBuffer(buf, BUFFER_LOCK_UNLOCK);
diff --git a/src/backend/access/nbtree/nbtree.c b/src/backend/access/nbtree/nbtree.c
index 8d4a587899c..d8e967402b2 100644
--- a/src/backend/access/nbtree/nbtree.c
+++ b/src/backend/access/nbtree/nbtree.c
@@ -1162,7 +1162,7 @@ btvacuumpage(BTVacState *vstate, BlockNumber scanblkno)
backtrack_to = P_NONE;
/* call vacuum_delay_point while not holding any buffer lock */
- vacuum_delay_point();
+ vacuum_delay_point(false);
/*
* We can't use _bt_getbuf() here because it always applies
diff --git a/src/backend/access/spgist/spgvacuum.c b/src/backend/access/spgist/spgvacuum.c
index 76fb0374c42..8188dbedce0 100644
--- a/src/backend/access/spgist/spgvacuum.c
+++ b/src/backend/access/spgist/spgvacuum.c
@@ -613,14 +613,16 @@ spgvacuumpage(spgBulkDeleteState *bds, BlockNumber blkno)
Relation index = bds->info->index;
Buffer buffer;
Page page;
+ bool was_empty;
/* call vacuum_delay_point while not holding any buffer lock */
- vacuum_delay_point();
+ vacuum_delay_point(false);
buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno,
RBM_NORMAL, bds->info->strategy);
LockBuffer(buffer, BUFFER_LOCK_EXCLUSIVE);
page = (Page) BufferGetPage(buffer);
+ was_empty = PageIsNew(page) || PageIsEmpty(page);
if (PageIsNew(page))
{
@@ -664,6 +666,8 @@ spgvacuumpage(spgBulkDeleteState *bds, BlockNumber blkno)
{
RecordFreeIndexPage(index, blkno);
bds->stats->pages_deleted++;
+ if (!was_empty)
+ bds->stats->pages_newly_deleted++;
}
else
{
@@ -694,7 +698,7 @@ spgprocesspending(spgBulkDeleteState *bds)
continue; /* ignore already-done items */
/* call vacuum_delay_point while not holding any buffer lock */
- vacuum_delay_point();
+ vacuum_delay_point(false);
/* examine the referenced page */
blkno = ItemPointerGetBlockNumber(&pitem->tid);
@@ -891,7 +895,6 @@ spgvacuumscan(spgBulkDeleteState *bds)
/* Report final stats */
bds->stats->num_pages = num_pages;
- bds->stats->pages_newly_deleted = bds->stats->pages_deleted;
bds->stats->pages_free = bds->stats->pages_deleted;
}
diff --git a/src/backend/commands/analyze.c b/src/backend/commands/analyze.c
index 2630c4943f8..c1026843596 100644
--- a/src/backend/commands/analyze.c
+++ b/src/backend/commands/analyze.c
@@ -539,6 +539,9 @@ do_analyze_rel(Relation onerel, VacuumParams *params,
save_sec_context | SECURITY_RESTRICTED_OPERATION);
save_nestlevel = NewGUCNestLevel();
+ /* Used for instrumentation and cumulative maintenance statistics. */
+ starttime = GetCurrentTimestamp();
+
/* measure elapsed time iff autovacuum logging requires it */
if (IsAutoVacuumWorkerProcess() && params->log_min_duration >= 0)
{
@@ -549,8 +552,6 @@ do_analyze_rel(Relation onerel, VacuumParams *params,
}
pg_rusage_init(&ru0);
- if (params->log_min_duration >= 0)
- starttime = GetCurrentTimestamp();
}
/*
@@ -1206,9 +1207,9 @@ do_analyze_rel(Relation onerel, VacuumParams *params,
*/
if (!inh)
pgstat_report_analyze(onerel, totalrows, totaldeadrows,
- (va_cols == NIL));
+ (va_cols == NIL), starttime);
else if (onerel->rd_rel->relkind == RELKIND_PARTITIONED_TABLE)
- pgstat_report_analyze(onerel, 0, 0, (va_cols == NIL));
+ pgstat_report_analyze(onerel, 0, 0, (va_cols == NIL), starttime);
/*
* If this isn't part of VACUUM ANALYZE, let index AMs do cleanup.
@@ -1406,7 +1407,7 @@ compute_index_stats(Relation onerel, double totalrows,
{
HeapTuple heapTuple = rows[rowno];
- vacuum_delay_point();
+ vacuum_delay_point(true);
/*
* Reset the per-tuple context each time, to reclaim any cruft
@@ -1826,7 +1827,7 @@ acquire_sample_rows(Relation onerel, int elevel,
prefetch_targblock = BlockSampler_Next(&prefetch_bs);
#endif
- vacuum_delay_point();
+ vacuum_delay_point(true);
block_accepted = table_scan_analyze_next_block(scan, targblock, vac_strategy);
@@ -3452,7 +3453,7 @@ compute_trivial_stats(VacAttrStatsP stats,
Datum value;
bool isnull;
- vacuum_delay_point();
+ vacuum_delay_point(true);
value = fetchfunc(stats, i, &isnull);
@@ -3574,7 +3575,7 @@ compute_distinct_stats(VacAttrStatsP stats,
int firstcount1,
j;
- vacuum_delay_point();
+ vacuum_delay_point(true);
value = fetchfunc(stats, i, &isnull);
@@ -3934,7 +3935,7 @@ compute_scalar_stats(VacAttrStatsP stats,
Datum value;
bool isnull;
- vacuum_delay_point();
+ vacuum_delay_point(true);
value = fetchfunc(stats, i, &isnull);
diff --git a/src/backend/commands/vacuum.c b/src/backend/commands/vacuum.c
index 65cf10bb833..40b36c7b9bc 100644
--- a/src/backend/commands/vacuum.c
+++ b/src/backend/commands/vacuum.c
@@ -104,6 +104,15 @@ static MemoryContext vac_context = NULL;
static BufferAccessStrategy vac_strategy;
+/*
+ * Cumulative time this process has spent in cost-based VACUUM delays, in
+ * microseconds. Consumers take differences around an operation. ANALYZE
+ * uses the same delay function but does not contribute to this accumulator.
+ * Updated only while track_cost_delay_timing is enabled.
+ */
+int64 VacuumDelayTime = 0;
+bool track_cost_delay_timing = false;
+
/*
* Variables for cost-based parallel vacuum. See comments atop
* compute_parallel_delay to understand how it works.
@@ -2985,7 +2994,7 @@ vac_close_indexes(int nindexes, Relation *Irel, LOCKMODE lockmode)
* typically once per page processed.
*/
void
-vacuum_delay_point(void)
+vacuum_delay_point(bool is_analyze)
{
double msec = 0;
@@ -3007,13 +3016,28 @@ vacuum_delay_point(void)
/* Nap if appropriate */
if (msec > 0)
{
+ instr_time delay_start;
+ instr_time delay_end;
+
if (msec > VacuumCostDelay * 4)
msec = VacuumCostDelay * 4;
pgstat_report_wait_start(WAIT_EVENT_VACUUM_DELAY);
+ if (track_cost_delay_timing && !is_analyze)
+ INSTR_TIME_SET_CURRENT(delay_start);
pg_usleep(msec * 1000);
pgstat_report_wait_end();
+ if (track_cost_delay_timing && !is_analyze)
+ {
+ int64 delay_us;
+
+ INSTR_TIME_SET_CURRENT(delay_end);
+ INSTR_TIME_SUBTRACT(delay_end, delay_start);
+ delay_us = (int64) INSTR_TIME_GET_MICROSEC(delay_end);
+ VacuumDelayTime += delay_us;
+ }
+
/*
* We don't want to ignore postmaster death during very long vacuums
* with vacuum_cost_delay configured. We can't use the usual
diff --git a/src/backend/commands/vacuum_ao.c b/src/backend/commands/vacuum_ao.c
index dff6ecf332d..72435bd03ba 100644
--- a/src/backend/commands/vacuum_ao.c
+++ b/src/backend/commands/vacuum_ao.c
@@ -321,7 +321,18 @@ ao_vacuum_rel_post_cleanup(Relation onerel, VacuumParams *params, BufferAccessSt
pgstat_report_vacuum(RelationGetRelid(onerel),
onerel->rd_rel->relisshared,
reltuples,
- deadtuples);
+ deadtuples,
+ vacrelstats->starttime,
+ vacrelstats->delay_time +
+ (VacuumDelayTime - vacrelstats->phase_start_delay),
+ false); /* AO itself has no failsafe mode. */
+
+ /*
+ * Remember what is left behind for the vacuum statistics, which
+ * ao_vacuum_rel() reports once this last phase is over.
+ */
+ vacrelstats->dead_tuples_left = (int64) deadtuples;
+ vacrelstats->total_file_segs = total_file_segs;
SIMPLE_FAULT_INJECTOR("vacuum_ao_post_cleanup_end");
}
@@ -429,11 +440,76 @@ init_vacrelstats()
old_context = MemoryContextSwitchTo(TopMemoryContext);
vacrelstats = (AOVacuumRelStats *) palloc0(sizeof(AOVacuumRelStats));
+ /* Time the run from its first phase in this worker. */
+ vacrelstats->starttime = GetCurrentTimestamp();
MemoryContextSwitchTo(old_context);
return vacrelstats;
}
+/*
+ * Report what the vacuuming of an append-optimized relation did to the
+ * statistics collector, the way lazy vacuum does for a heap relation.
+ *
+ * The counters that describe heap pages have no counterpart here: an AO
+ * relation has no heap visibility map and nothing to freeze, so
+ * pages_frozen and pages_all_visible stay zero. Nor is there a relfrozenxid
+ * of its own to reach the freeze table age -- it is always invalid, the
+ * auxiliary heap relations being vacuumed and frozen on their own -- so the
+ * freeze_age_vacuum_count stays zero as well. What the heap reports as
+ * truncated pages is the space compaction freed, measured in blocks of the
+ * segment files it dropped or truncated.
+ *
+ * The indexes report themselves, from vacuum_appendonly_index() and scan_index().
+ */
+static void
+ao_report_vacuum_stats(Relation aorel, AOVacuumRelStats *vacrelstats)
+{
+ PgStat_VacuumStats vacstats;
+
+ Assert(pgstat_track_vacuum_statistics);
+
+ MemSet(&vacstats, 0, sizeof(vacstats));
+ vacstats.tuples_deleted = (PgStat_Counter) vacrelstats->num_dead_tuples;
+ vacstats.dead_tuples = (PgStat_Counter) vacrelstats->dead_tuples_left;
+ /* Keep the conversion 64-bit: aggregate AO files can exceed BlockNumber. */
+ vacstats.bytes_removed = vacrelstats->nbytes_truncated;
+ vacstats.pages_deleted = vacrelstats->nbytes_truncated / BLCKSZ +
+ (vacrelstats->nbytes_truncated % BLCKSZ != 0);
+ vacstats.total_file_segs = vacrelstats->total_file_segs;
+
+ pgstat_report_vacstats(RelationGetRelid(aorel),
+ aorel->rd_rel->relisshared,
+ false,
+ &vacstats);
+}
+
+/* Report index work, including cleanup runs with no obsolete AO segments. */
+static void
+ao_report_index_vacuum_stats(Relation indexRelation,
+ IndexBulkDeleteResult *stats)
+{
+ PgStat_VacuumStats vacstats;
+
+ Assert(pgstat_track_vacuum_statistics);
+
+ MemSet(&vacstats, 0, sizeof(vacstats));
+ if (stats)
+ {
+ vacstats.tuples_deleted = (PgStat_Counter) stats->tuples_removed;
+ vacstats.pages_deleted = (PgStat_Counter) stats->pages_newly_deleted;
+ /* deleted pages that are not yet reusable still hold dead entries */
+ if (stats->pages_deleted > stats->pages_free)
+ vacstats.dead_pages =
+ (PgStat_Counter) (stats->pages_deleted - stats->pages_free);
+ }
+
+ pgstat_report_vacstats(RelationGetRelid(indexRelation),
+ indexRelation->rd_rel->relisshared,
+ true,
+ &vacstats);
+}
+
/*
* ao_vacuum_rel()
*
@@ -443,6 +519,10 @@ void
ao_vacuum_rel(Relation rel, VacuumParams *params, BufferAccessStrategy bstrategy)
{
static AOVacuumRelStats *vacrelstats = NULL;
+ instr_time phasestart;
+ instr_time phaseend;
+ int64 startdelaytime;
+
Assert(RelationStorageIsAO(rel));
Assert(params != NULL);
@@ -469,20 +549,56 @@ ao_vacuum_rel(Relation rel, VacuumParams *params, BufferAccessStrategy bstrategy
/*
* Do the actual work --- either FULL or "lazy" vacuum
+ *
+ * Each phase is timed for the vacuum statistics, which are reported for
+ * the relation once the last phase is done. The phases run in separate
+ * transactions and may even end up in different vacuum workers; when that
+ * happens vacrelstats is reset above, and the statistics report describes
+ * the phases this worker did.
*/
+ INSTR_TIME_SET_CURRENT(phasestart);
+ startdelaytime = VacuumDelayTime;
+ vacrelstats->phase_start_delay = startdelaytime;
+
if (ao_vacuum_phase == VACOPT_AO_PRE_CLEANUP_PHASE)
ao_vacuum_rel_pre_cleanup(rel, params, bstrategy, vacrelstats);
else if (ao_vacuum_phase == VACOPT_AO_COMPACT_PHASE)
ao_vacuum_rel_compact(rel, params, bstrategy, vacrelstats);
else if (ao_vacuum_phase == VACOPT_AO_POST_CLEANUP_PHASE)
- {
ao_vacuum_rel_post_cleanup(rel, params, bstrategy, vacrelstats);
- pgstat_progress_end_command();
- cleanup_vacrelstats(&vacrelstats);
- }
else
/* Do nothing here, we will launch the stages later */
Assert(ao_vacuum_phase == 0);
+
+ INSTR_TIME_SET_CURRENT(phaseend);
+ INSTR_TIME_SUBTRACT(phaseend, phasestart);
+ vacrelstats->vacuum_time += (int64) INSTR_TIME_GET_MICROSEC(phaseend);
+ vacrelstats->delay_time += VacuumDelayTime - startdelaytime;
+
+ if (ao_vacuum_phase == VACOPT_AO_POST_CLEANUP_PHASE)
+ {
+ int elevel = (params->options & VACOPT_VERBOSE) ? INFO : DEBUG2;
+
+ if (Gp_role == GP_ROLE_DISPATCH)
+ elevel = DEBUG2;
+
+ ereport(elevel,
+ (errmsg("append-optimized table \"%s\": vacuum statistics",
+ RelationGetRelationName(rel)),
+ errdetail("%lld dead tuples remain.\n"
+ "%lld bytes truncated; %lld file segments remain.\n"
+ "elapsed: %.3f ms, cost-based delay: %.3f ms",
+ (long long) vacrelstats->dead_tuples_left,
+ (long long) vacrelstats->nbytes_truncated,
+ (long long) vacrelstats->total_file_segs,
+ vacrelstats->vacuum_time / 1000.0,
+ vacrelstats->delay_time / 1000.0)));
+
+ if (pgstat_track_vacuum_statistics)
+ ao_report_vacuum_stats(rel, vacrelstats);
+ pgstat_progress_end_command();
+ cleanup_vacrelstats(&vacrelstats);
+ }
}
/*
@@ -633,10 +749,15 @@ vacuum_appendonly_index(Relation indexRelation,
IndexBulkDeleteResult *stats;
IndexVacuumInfo ivinfo = {0};
PGRUsage ru0;
+ instr_time starttime;
+ instr_time endtime;
+ int64 startdelaytime;
Assert(RelationIsValid(indexRelation));
pg_rusage_init(&ru0);
+ INSTR_TIME_SET_CURRENT(starttime);
+ startdelaytime = VacuumDelayTime;
ivinfo.index = indexRelation;
ivinfo.analyze_only = false;
@@ -661,6 +782,17 @@ vacuum_appendonly_index(Relation indexRelation,
/* Do post-VACUUM cleanup */
stats = index_vacuum_cleanup(&ivinfo, stats);
+ INSTR_TIME_SET_CURRENT(endtime);
+ INSTR_TIME_SUBTRACT(endtime, starttime);
+
+ /* Ordinary timing follows track_counts, independently of work counters. */
+ pgstat_report_index_vacuum_time(indexRelation,
+ (PgStat_Counter) INSTR_TIME_GET_MICROSEC(endtime),
+ VacuumDelayTime - startdelaytime,
+ IsAutoVacuumWorkerProcess());
+ if (pgstat_track_vacuum_statistics)
+ ao_report_index_vacuum_stats(indexRelation, stats);
+
if (!stats)
return;
@@ -685,9 +817,11 @@ vacuum_appendonly_index(Relation indexRelation,
stats->num_pages),
errdetail("%.0f index row versions were removed.\n"
"%u index pages have been deleted, %u are currently reusable.\n"
+ "cost-based delay: %.3f ms\n"
"%s.",
stats->tuples_removed,
stats->pages_deleted, stats->pages_free,
+ (VacuumDelayTime - startdelaytime) / 1000.0,
pg_rusage_show(&ru0))));
pfree(stats);
@@ -806,8 +940,13 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac
IndexBulkDeleteResult *stats;
IndexVacuumInfo ivinfo = {0};
PGRUsage ru0;
+ instr_time starttime;
+ instr_time endtime;
+ int64 startdelaytime;
pg_rusage_init(&ru0);
+ INSTR_TIME_SET_CURRENT(starttime);
+ startdelaytime = VacuumDelayTime;
ivinfo.index = indrel;
ivinfo.analyze_only = false;
@@ -824,6 +963,17 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac
/* Do post-VACUUM cleanup */
stats = index_vacuum_cleanup(&ivinfo, NULL);
+ INSTR_TIME_SET_CURRENT(endtime);
+ INSTR_TIME_SUBTRACT(endtime, starttime);
+
+ /* Ordinary timing follows track_counts, independently of work counters. */
+ pgstat_report_index_vacuum_time(indrel,
+ (PgStat_Counter) INSTR_TIME_GET_MICROSEC(endtime),
+ VacuumDelayTime - startdelaytime,
+ IsAutoVacuumWorkerProcess());
+ if (pgstat_track_vacuum_statistics)
+ ao_report_index_vacuum_stats(indrel, stats);
+
if (!stats)
return;
@@ -847,8 +997,10 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac
stats->num_index_tuples,
stats->num_pages),
errdetail("%u index pages have been deleted, %u are currently reusable.\n"
+ "cost-based delay: %.3f ms\n"
"%s.",
stats->pages_deleted, stats->pages_free,
+ (VacuumDelayTime - startdelaytime) / 1000.0,
pg_rusage_show(&ru0))));
pfree(stats);
diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c
index 309a101fe02..b6f3592ff9a 100644
--- a/src/backend/postmaster/pgstat.c
+++ b/src/backend/postmaster/pgstat.c
@@ -39,6 +39,7 @@
#include "access/twophase_rmgr.h"
#include "access/xact.h"
#include "access/xlog.h"
+#include "catalog/catalog.h"
#include "catalog/pg_database.h"
#include "catalog/pg_proc.h"
#include "executor/instrument.h"
@@ -126,6 +127,7 @@
* ----------
*/
bool pgstat_track_counts = false;
+bool pgstat_track_vacuum_statistics = false;
int pgstat_track_functions = TRACK_FUNC_OFF;
bool pgstat_collect_queuelevel = false;
@@ -258,6 +260,10 @@ static PgStat_SubXactStatus *pgStatXactStack = NULL;
static int pgStatXactCommit = 0;
static int pgStatXactRollback = 0;
+
+/* Only incremented in error callbacks; sent by pgstat_report_stat(). */
+static PgStat_Counter pgStatVacuumErrors = 0;
+static PgStat_Counter pgStatSharedVacuumErrors = 0;
PgStat_Counter pgStatBlockReadTime = 0;
PgStat_Counter pgStatBlockWriteTime = 0;
static PgStat_Counter pgLastSessionReportTime = 0;
@@ -372,6 +378,8 @@ static void pgstat_recv_resetslrucounter(PgStat_MsgResetslrucounter *msg, int le
static void pgstat_recv_resetreplslotcounter(PgStat_MsgResetreplslotcounter *msg, int len);
static void pgstat_recv_autovac(PgStat_MsgAutovacStart *msg, int len);
static void pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len);
+static void pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len);
+static void pgstat_recv_resetvacstats(PgStat_MsgResetVacstats *msg, int len);
static void pgstat_recv_analyze(PgStat_MsgAnalyze *msg, int len);
static void pgstat_recv_archiver(PgStat_MsgArchiver *msg, int len);
static void pgstat_recv_queuestat(PgStat_MsgQueuestat *msg, int len); /* GPDB */
@@ -893,6 +901,7 @@ pgstat_report_stat(bool disconnect)
*/
if ((pgStatTabList == NULL || pgStatTabList->tsa_used == 0) &&
pgStatXactCommit == 0 && pgStatXactRollback == 0 &&
+ pgStatVacuumErrors == 0 && pgStatSharedVacuumErrors == 0 &&
pgWalUsage.wal_records == prevWalUsage.wal_records &&
WalStats.m_wal_write == 0 && WalStats.m_wal_sync == 0 &&
!have_function_stats && !disconnect)
@@ -975,13 +984,14 @@ pgstat_report_stat(bool disconnect)
/*
* Send partial messages. Make sure that any pending xact commit/abort
- * and connection stats get counted, even if there are no table stats to
- * send.
+ * counts, connection stats and interrupted vacuums get counted, even if
+ * there are no table stats to send.
*/
if (regular_msg.m_nentries > 0 ||
- pgStatXactCommit > 0 || pgStatXactRollback > 0 || disconnect)
+ pgStatXactCommit > 0 || pgStatXactRollback > 0 ||
+ pgStatVacuumErrors > 0 || disconnect)
pgstat_send_tabstat(®ular_msg, now);
- if (shared_msg.m_nentries > 0)
+ if (shared_msg.m_nentries > 0 || pgStatSharedVacuumErrors > 0)
pgstat_send_tabstat(&shared_msg, now);
/* Now, send function statistics */
@@ -1008,13 +1018,15 @@ pgstat_send_tabstat(PgStat_MsgTabstat *tsmsg, TimestampTz now)
return;
/*
- * Report and reset accumulated xact commit/rollback and I/O timings
- * whenever we send a normal tabstat message
+ * Report and reset accumulated xact commit/rollback, I/O timings and
+ * interrupted vacuums whenever we send a normal tabstat message.
*/
if (OidIsValid(tsmsg->m_databaseid))
{
tsmsg->m_xact_commit = pgStatXactCommit;
tsmsg->m_xact_rollback = pgStatXactRollback;
+ tsmsg->m_vacuum_interrupt_count = pgStatVacuumErrors;
+ pgStatVacuumErrors = 0;
tsmsg->m_block_read_time = pgStatBlockReadTime;
tsmsg->m_block_write_time = pgStatBlockWriteTime;
@@ -1050,6 +1062,8 @@ pgstat_send_tabstat(PgStat_MsgTabstat *tsmsg, TimestampTz now)
{
tsmsg->m_xact_commit = 0;
tsmsg->m_xact_rollback = 0;
+ tsmsg->m_vacuum_interrupt_count = pgStatSharedVacuumErrors;
+ pgStatSharedVacuumErrors = 0;
tsmsg->m_block_read_time = 0;
tsmsg->m_block_write_time = 0;
tsmsg->m_session_time = 0;
@@ -1589,7 +1603,9 @@ pgstat_report_autovac(Oid dboid)
*/
void
pgstat_report_vacuum(Oid tableoid, bool shared,
- PgStat_Counter livetuples, PgStat_Counter deadtuples)
+ PgStat_Counter livetuples, PgStat_Counter deadtuples,
+ TimestampTz starttime, PgStat_Counter delaytime,
+ bool failsafe)
{
PgStat_MsgVacuum msg;
@@ -1600,12 +1616,112 @@ pgstat_report_vacuum(Oid tableoid, bool shared,
msg.m_databaseid = shared ? InvalidOid : MyDatabaseId;
msg.m_tableoid = tableoid;
msg.m_autovacuum = IsAutoVacuumWorkerProcess();
+ msg.m_isindex = false;
+ msg.m_failsafe = failsafe;
+ msg.m_delaytime = delaytime;
msg.m_vacuumtime = GetCurrentTimestamp();
+ msg.m_elapsedtime = Max(msg.m_vacuumtime - starttime, 0);
msg.m_live_tuples = livetuples;
msg.m_dead_tuples = deadtuples;
pgstat_send(&msg, sizeof(msg));
}
+/*
+ * Count a heap vacuum interrupted by ERROR. The caller is an error context
+ * callback, possibly running while a lock is held. Do not allocate memory,
+ * acquire locks or send messages here. Like transaction counts, these local
+ * counters are sent by pgstat_report_stat() outside a transaction. Shared
+ * relations belong to the InvalidOid database entry.
+ */
+void
+pgstat_count_vacuum_error(bool shared)
+{
+ if (!pgstat_track_counts)
+ return;
+
+ if (shared)
+ pgStatSharedVacuumErrors++;
+ else
+ pgStatVacuumErrors++;
+}
+
+/* Report an index pass without changing table estimates or vacuum counts. */
+void
+pgstat_report_index_vacuum_time(Relation rel, PgStat_Counter elapsedtime,
+ PgStat_Counter delaytime, bool is_autovacuum)
+{
+ PgStat_MsgVacuum msg;
+
+ if (pgStatSock == PGINVALID_SOCKET || !pgstat_track_counts)
+ return;
+
+ MemSet(&msg, 0, sizeof(msg));
+ pgstat_setheader(&msg.m_hdr, PGSTAT_MTYPE_VACUUM);
+ msg.m_databaseid = rel->rd_rel->relisshared ? InvalidOid : MyDatabaseId;
+ msg.m_tableoid = RelationGetRelid(rel);
+ msg.m_autovacuum = is_autovacuum;
+ msg.m_isindex = true;
+ msg.m_elapsedtime = elapsedtime;
+ msg.m_delaytime = delaytime;
+ pgstat_send(&msg, sizeof(msg));
+}
+
+/* ---------
+ * pgstat_report_vacstats() -
+ *
+ * Tell the collector about the counters accumulated while vacuuming a
+ * relation (a table or an index). isindex tells which one, since only
+ * the tables count towards the per-database totals.
+ * ---------
+ */
+void
+pgstat_report_vacstats(Oid tableoid, bool shared, bool isindex,
+ const PgStat_VacuumStats *stats)
+{
+ PgStat_MsgVacstats msg;
+
+ if (pgStatSock == PGINVALID_SOCKET || !pgstat_track_counts ||
+ !pgstat_track_vacuum_statistics)
+ return;
+
+ pgstat_setheader(&msg.m_hdr, PGSTAT_MTYPE_VACSTATS);
+ msg.m_databaseid = shared ? InvalidOid : MyDatabaseId;
+ msg.m_tableoid = tableoid;
+ msg.m_isindex = isindex;
+ msg.m_stats = *stats;
+ pgstat_send(&msg, sizeof(msg));
+}
+
+/* ----------
+ * pgstat_reset_vacuum_stats() -
+ *
+ * Tell the collector to throw away the vacuum counters of one relation of
+ * this database, or of all of them when resetall is true.
+ * ----------
+ */
+void
+pgstat_reset_vacuum_stats(Oid relid, bool resetall)
+{
+ PgStat_MsgResetVacstats msg;
+
+ /* An invalid relation OID must never turn into a database-wide reset. */
+ if (!resetall && !OidIsValid(relid))
+ ereport(ERROR,
+ (errcode(ERRCODE_INVALID_PARAMETER_VALUE),
+ errmsg("invalid relation OID: %u", relid)));
+ Assert(!resetall || !OidIsValid(relid));
+
+ if (pgStatSock == PGINVALID_SOCKET)
+ return;
+
+ pgstat_setheader(&msg.m_hdr, PGSTAT_MTYPE_RESETVACSTATS);
+ msg.m_databaseid = !resetall && IsSharedRelation(relid) ?
+ InvalidOid : MyDatabaseId;
+ msg.m_objectid = relid;
+ msg.m_resetall = resetall;
+ pgstat_send(&msg, sizeof(msg));
+}
+
/* --------
* pgstat_report_analyze() -
*
@@ -1618,7 +1734,7 @@ pgstat_report_vacuum(Oid tableoid, bool shared,
void
pgstat_report_analyze(Relation rel,
PgStat_Counter livetuples, PgStat_Counter deadtuples,
- bool resetcounter)
+ bool resetcounter, TimestampTz starttime)
{
PgStat_MsgAnalyze msg;
@@ -1660,6 +1776,7 @@ pgstat_report_analyze(Relation rel,
msg.m_autovacuum = IsAutoVacuumWorkerProcess();
msg.m_resetcounter = resetcounter;
msg.m_analyzetime = GetCurrentTimestamp();
+ msg.m_elapsedtime = Max(msg.m_analyzetime - starttime, 0);
msg.m_live_tuples = livetuples;
msg.m_dead_tuples = deadtuples;
pgstat_send(&msg, sizeof(msg));
@@ -2805,6 +2922,25 @@ pgstat_fetch_stat_tabentry(Oid relid)
}
+/* ----------
+ * pgstat_fetch_stat_vacuum_stats() -
+ *
+ * Return the vacuum counters available for a relation, or NULL.
+ * ----------
+ */
+PgStat_VacuumStats *
+pgstat_fetch_stat_vacuum_stats(Oid relid)
+{
+ PgStat_StatTabEntry *tabentry;
+
+ if (!pgstat_track_vacuum_statistics)
+ return NULL;
+
+ tabentry = pgstat_fetch_stat_tabentry(relid);
+ return tabentry ? &tabentry->vacuum_stats : NULL;
+}
+
+
/* ----------
* pgstat_fetch_stat_funcentry() -
*
@@ -3677,6 +3813,14 @@ PgstatCollectorMain(int argc, char *argv[])
pgstat_recv_vacuum(&msg.msg_vacuum, len);
break;
+ case PGSTAT_MTYPE_VACSTATS:
+ pgstat_recv_vacstats(&msg.msg_vacstats, len);
+ break;
+
+ case PGSTAT_MTYPE_RESETVACSTATS:
+ pgstat_recv_resetvacstats(&msg.msg_resetvacstats, len);
+ break;
+
case PGSTAT_MTYPE_ANALYZE:
pgstat_recv_analyze(&msg.msg_analyze, len);
break;
@@ -3820,12 +3964,23 @@ reset_dbentry_counters(PgStat_StatDBEntry *dbentry)
dbentry->n_sessions_abandoned = 0;
dbentry->n_sessions_fatal = 0;
dbentry->n_sessions_killed = 0;
+ dbentry->total_vacuum_time = 0;
+ dbentry->total_autovacuum_time = 0;
+ dbentry->total_vacuum_delay_time = 0;
+ dbentry->total_autovacuum_delay_time = 0;
+ dbentry->vacuum_failsafe_count = 0;
+ dbentry->vacuum_interrupt_count = 0;
+
+ dbentry->n_frozen_page_marks_cleared = 0;
+ dbentry->n_visible_page_marks_cleared = 0;
+ if (pgstat_track_vacuum_statistics)
+ MemSet(&dbentry->n_vacuum_stats, 0, sizeof(dbentry->n_vacuum_stats));
dbentry->stat_reset_timestamp = GetCurrentTimestamp();
dbentry->stats_timestamp = 0;
hash_ctl.keysize = sizeof(Oid);
- hash_ctl.entrysize = sizeof(PgStat_StatTabEntry);
+ hash_ctl.entrysize = PGSTAT_TAB_ENTRY_SIZE;
dbentry->tables = hash_create("Per-database table",
PGSTAT_TAB_HASH_SIZE,
&hash_ctl,
@@ -3914,6 +4069,17 @@ pgstat_get_tab_entry(PgStat_StatDBEntry *dbentry, Oid tableoid, bool create)
result->analyze_count = 0;
result->autovac_analyze_timestamp = 0;
result->autovac_analyze_count = 0;
+ result->total_vacuum_time = 0;
+ result->total_autovacuum_time = 0;
+ result->total_analyze_time = 0;
+ result->total_autoanalyze_time = 0;
+ result->total_vacuum_delay_time = 0;
+ result->total_autovacuum_delay_time = 0;
+ result->vacuum_failsafe_count = 0;
+ result->frozen_page_marks_cleared = 0;
+ result->visible_page_marks_cleared = 0;
+ if (pgstat_track_vacuum_statistics)
+ MemSet(&result->vacuum_stats, 0, sizeof(result->vacuum_stats));
}
return result;
@@ -4020,9 +4186,14 @@ pgstat_write_statsfiles(bool permanent, bool allDbs)
* Write out the DB entry. We don't write the tables or functions
* pointers, since they're of no use to any other process.
*/
- fputc('D', fpout);
+ fputc(pgstat_track_vacuum_statistics ? 'd' : 'D', fpout);
rc = fwrite(dbentry, offsetof(PgStat_StatDBEntry, tables), 1, fpout);
(void) rc; /* we'll check for error with ferror */
+ if (pgstat_track_vacuum_statistics)
+ {
+ rc = fwrite(&dbentry->n_vacuum_stats, sizeof(PgStat_VacuumStats), 1, fpout);
+ (void) rc;
+ }
}
/*
@@ -4169,8 +4340,8 @@ pgstat_write_db_statsfile(PgStat_StatDBEntry *dbentry, bool permanent)
hash_seq_init(&tstat, dbentry->tables);
while ((tabentry = (PgStat_StatTabEntry *) hash_seq_search(&tstat)) != NULL)
{
- fputc('T', fpout);
- rc = fwrite(tabentry, sizeof(PgStat_StatTabEntry), 1, fpout);
+ fputc(pgstat_track_vacuum_statistics ? 't' : 'T', fpout);
+ rc = fwrite(tabentry, PGSTAT_TAB_ENTRY_SIZE, 1, fpout);
(void) rc; /* we'll check for error with ferror */
}
@@ -4256,6 +4427,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep)
HTAB *dbhash;
FILE *fpin;
int32 format_id;
+ int record_type;
bool found;
const char *statfile = permanent ? PGSTAT_STAT_PERMANENT_FILENAME : pgstat_stat_filename;
int i;
@@ -4273,7 +4445,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep)
* Create the DB hashtable
*/
hash_ctl.keysize = sizeof(Oid);
- hash_ctl.entrysize = sizeof(PgStat_StatDBEntry);
+ hash_ctl.entrysize = PGSTAT_DB_ENTRY_SIZE;
hash_ctl.hcxt = pgStatLocalContext;
dbhash = hash_create("Databases hash", PGSTAT_DB_HASH_SIZE, &hash_ctl,
HASH_ELEM | HASH_BLOBS | HASH_CONTEXT);
@@ -4403,13 +4575,15 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep)
*/
for (;;)
{
- switch (fgetc(fpin))
+ switch (record_type = fgetc(fpin))
{
/*
- * 'D' A PgStat_StatDBEntry struct describing a database
- * follows.
+ * 'D' Ordinary database counters follow.
+ * 'd' The same, followed by a PgStat_VacuumStats block.
*/
case 'D':
+ case 'd':
+ MemSet(&dbbuf, 0, sizeof(dbbuf));
if (fread(&dbbuf, 1, offsetof(PgStat_StatDBEntry, tables),
fpin) != offsetof(PgStat_StatDBEntry, tables))
{
@@ -4419,6 +4593,15 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep)
goto done;
}
+ if (record_type == 'd' &&
+ fread(&dbbuf.n_vacuum_stats, 1, sizeof(PgStat_VacuumStats),
+ fpin) != sizeof(PgStat_VacuumStats))
+ {
+ ereport(pgStatRunningInCollector ? LOG : WARNING,
+ (errmsg("corrupted statistics file \"%s\"", statfile)));
+ goto done;
+ }
+
/*
* Add to the DB hash
*/
@@ -4434,7 +4617,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep)
goto done;
}
- memcpy(dbentry, &dbbuf, sizeof(PgStat_StatDBEntry));
+ memcpy(dbentry, &dbbuf, PGSTAT_DB_ENTRY_SIZE);
dbentry->tables = NULL;
dbentry->functions = NULL;
@@ -4459,7 +4642,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep)
}
hash_ctl.keysize = sizeof(Oid);
- hash_ctl.entrysize = sizeof(PgStat_StatTabEntry);
+ hash_ctl.entrysize = PGSTAT_TAB_ENTRY_SIZE;
hash_ctl.hcxt = pgStatLocalContext;
dbentry->tables = hash_create("Per-database table",
PGSTAT_TAB_HASH_SIZE,
@@ -4608,6 +4791,8 @@ pgstat_read_db_statsfile(Oid databaseid, HTAB *tabhash, HTAB *funchash,
PgStat_StatFuncEntry *funcentry;
FILE *fpin;
int32 format_id;
+ int record_type;
+ size_t tabsize;
bool found;
char statfile[MAXPGPATH];
@@ -4649,14 +4834,18 @@ pgstat_read_db_statsfile(Oid databaseid, HTAB *tabhash, HTAB *funchash,
*/
for (;;)
{
- switch (fgetc(fpin))
+ switch (record_type = fgetc(fpin))
{
/*
- * 'T' A PgStat_StatTabEntry follows.
+ * 'T' An ordinary table entry follows.
+ * 't' The entry also includes its vacuum counters.
*/
case 'T':
- if (fread(&tabbuf, 1, sizeof(PgStat_StatTabEntry),
- fpin) != sizeof(PgStat_StatTabEntry))
+ case 't':
+ tabsize = record_type == 't' ? sizeof(PgStat_StatTabEntry) :
+ offsetof(PgStat_StatTabEntry, vacuum_stats);
+ MemSet(&tabbuf, 0, sizeof(tabbuf));
+ if (fread(&tabbuf, 1, tabsize, fpin) != tabsize)
{
ereport(pgStatRunningInCollector ? LOG : WARNING,
(errmsg("corrupted statistics file \"%s\"",
@@ -4682,7 +4871,7 @@ pgstat_read_db_statsfile(Oid databaseid, HTAB *tabhash, HTAB *funchash,
goto done;
}
- memcpy(tabentry, &tabbuf, sizeof(tabbuf));
+ memcpy(tabentry, &tabbuf, PGSTAT_TAB_ENTRY_SIZE);
break;
/*
@@ -4774,6 +4963,7 @@ pgstat_read_db_statsfile_timestamp(Oid databaseid, bool permanent,
PgStat_StatReplSlotEntry myReplSlotStats;
FILE *fpin;
int32 format_id;
+ int record_type;
const char *statfile = permanent ? PGSTAT_STAT_PERMANENT_FILENAME : pgstat_stat_filename;
/*
@@ -4857,13 +5047,15 @@ pgstat_read_db_statsfile_timestamp(Oid databaseid, bool permanent,
*/
for (;;)
{
- switch (fgetc(fpin))
+ switch (record_type = fgetc(fpin))
{
/*
- * 'D' A PgStat_StatDBEntry struct describing a database
- * follows.
+ * 'D' Ordinary database counters follow.
+ * 'd' The same, followed by a PgStat_VacuumStats block.
*/
case 'D':
+ case 'd':
+ MemSet(&dbentry, 0, sizeof(dbentry));
if (fread(&dbentry, 1, offsetof(PgStat_StatDBEntry, tables),
fpin) != offsetof(PgStat_StatDBEntry, tables))
{
@@ -4874,6 +5066,16 @@ pgstat_read_db_statsfile_timestamp(Oid databaseid, bool permanent,
return false;
}
+ if (record_type == 'd' &&
+ fread(&dbentry.n_vacuum_stats, 1, sizeof(PgStat_VacuumStats),
+ fpin) != sizeof(PgStat_VacuumStats))
+ {
+ ereport(pgStatRunningInCollector ? LOG : WARNING,
+ (errmsg("corrupted statistics file \"%s\"", statfile)));
+ FreeFile(fpin);
+ return false;
+ }
+
/*
* If this is the DB we're looking for, save its timestamp and
* we're done.
@@ -5229,6 +5431,7 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len)
*/
dbentry->n_xact_commit += (PgStat_Counter) (msg->m_xact_commit);
dbentry->n_xact_rollback += (PgStat_Counter) (msg->m_xact_rollback);
+ dbentry->vacuum_interrupt_count += msg->m_vacuum_interrupt_count;
dbentry->n_block_read_time += msg->m_block_read_time;
dbentry->n_block_write_time += msg->m_block_write_time;
@@ -5275,6 +5478,17 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len)
tabentry->analyze_count = 0;
tabentry->autovac_analyze_timestamp = 0;
tabentry->autovac_analyze_count = 0;
+ tabentry->total_vacuum_time = 0;
+ tabentry->total_autovacuum_time = 0;
+ tabentry->total_analyze_time = 0;
+ tabentry->total_autoanalyze_time = 0;
+ tabentry->total_vacuum_delay_time = 0;
+ tabentry->total_autovacuum_delay_time = 0;
+ tabentry->vacuum_failsafe_count = 0;
+ tabentry->frozen_page_marks_cleared = 0;
+ tabentry->visible_page_marks_cleared = 0;
+ if (pgstat_track_vacuum_statistics)
+ MemSet(&tabentry->vacuum_stats, 0, sizeof(tabentry->vacuum_stats));
}
else
{
@@ -5318,6 +5532,14 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len)
dbentry->n_tuples_deleted += tabmsg->t_counts.t_tuples_deleted;
dbentry->n_blocks_fetched += tabmsg->t_counts.t_blocks_fetched;
dbentry->n_blocks_hit += tabmsg->t_counts.t_blocks_hit;
+ tabentry->frozen_page_marks_cleared +=
+ tabmsg->t_counts.t_frozen_page_marks_cleared;
+ tabentry->visible_page_marks_cleared +=
+ tabmsg->t_counts.t_visible_page_marks_cleared;
+ dbentry->n_frozen_page_marks_cleared +=
+ tabmsg->t_counts.t_frozen_page_marks_cleared;
+ dbentry->n_visible_page_marks_cleared +=
+ tabmsg->t_counts.t_visible_page_marks_cleared;
}
}
@@ -5605,6 +5827,38 @@ pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len)
tabentry = pgstat_get_tab_entry(dbentry, msg->m_tableoid, true);
+ if (msg->m_autovacuum)
+ {
+ tabentry->total_autovacuum_time += msg->m_elapsedtime;
+ tabentry->total_autovacuum_delay_time += msg->m_delaytime;
+ }
+ else
+ {
+ tabentry->total_vacuum_time += msg->m_elapsedtime;
+ tabentry->total_vacuum_delay_time += msg->m_delaytime;
+ }
+
+ /* Index passes are already included in the owning table's elapsed time. */
+ if (msg->m_isindex)
+ return;
+
+ if (msg->m_failsafe)
+ {
+ tabentry->vacuum_failsafe_count++;
+ dbentry->vacuum_failsafe_count++;
+ }
+
+ if (msg->m_autovacuum)
+ {
+ dbentry->total_autovacuum_time += msg->m_elapsedtime;
+ dbentry->total_autovacuum_delay_time += msg->m_delaytime;
+ }
+ else
+ {
+ dbentry->total_vacuum_time += msg->m_elapsedtime;
+ dbentry->total_vacuum_delay_time += msg->m_delaytime;
+ }
+
tabentry->n_live_tuples = msg->m_live_tuples;
tabentry->n_dead_tuples = msg->m_dead_tuples;
@@ -5632,6 +5886,120 @@ pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len)
}
}
+/* ----------
+ * pgstat_recv_vacstats() -
+ *
+ * Process a VACSTATS message: accumulate the vacuum counters into the
+ * relation's entry and, for a table, into the per-database totals.
+ * ----------
+ */
+static void
+pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len)
+{
+ PgStat_StatDBEntry *dbentry;
+ PgStat_VacuumStats *vacstats;
+
+ if (!pgstat_track_vacuum_statistics)
+ return;
+
+ dbentry = pgstat_get_db_entry(msg->m_databaseid, true);
+ vacstats = &pgstat_get_tab_entry(dbentry, msg->m_tableoid, true)->vacuum_stats;
+
+ vacstats->tuples_deleted += msg->m_stats.tuples_deleted;
+ vacstats->dead_tuples += msg->m_stats.dead_tuples;
+ vacstats->pages_deleted += msg->m_stats.pages_deleted;
+ vacstats->bytes_removed += msg->m_stats.bytes_removed;
+ /* Relation state is replaced, never added to database totals. */
+ vacstats->total_file_segs = msg->m_stats.total_file_segs;
+ vacstats->dead_pages += msg->m_stats.dead_pages;
+ vacstats->pages_frozen += msg->m_stats.pages_frozen;
+ vacstats->pages_all_visible += msg->m_stats.pages_all_visible;
+ vacstats->freeze_age_vacuum_count +=
+ msg->m_stats.freeze_age_vacuum_count;
+
+ /*
+ * The per-database totals describe what vacuum did to the tables. An
+ * index is vacuumed as a part of its table, and the time it took is
+ * already accounted for in the table's own report, so adding the index
+ * counters here would count that work twice.
+ */
+ if (msg->m_isindex)
+ return;
+
+ dbentry->n_vacuum_stats.tuples_deleted += msg->m_stats.tuples_deleted;
+ dbentry->n_vacuum_stats.dead_tuples += msg->m_stats.dead_tuples;
+ dbentry->n_vacuum_stats.pages_deleted += msg->m_stats.pages_deleted;
+ dbentry->n_vacuum_stats.bytes_removed += msg->m_stats.bytes_removed;
+ dbentry->n_vacuum_stats.dead_pages += msg->m_stats.dead_pages;
+ dbentry->n_vacuum_stats.pages_frozen += msg->m_stats.pages_frozen;
+ dbentry->n_vacuum_stats.pages_all_visible += msg->m_stats.pages_all_visible;
+ dbentry->n_vacuum_stats.freeze_age_vacuum_count +=
+ msg->m_stats.freeze_age_vacuum_count;
+}
+
+/* ----------
+ * pgstat_recv_resetvacstats() -
+ *
+ * Throw away the vacuum counters of one relation, or of the whole database
+ * when no relation is given. This includes the VM revision counters;
+ * ordinary statistics are left alone.
+ * ----------
+ */
+static void
+pgstat_recv_resetvacstats(PgStat_MsgResetVacstats *msg, int len)
+{
+ PgStat_StatDBEntry *dbentry;
+ PgStat_StatTabEntry *tabentry;
+ HASH_SEQ_STATUS hstat;
+
+ dbentry = pgstat_get_db_entry(msg->m_databaseid, false);
+ if (!dbentry)
+ return;
+
+ if (!msg->m_resetall)
+ {
+ tabentry = pgstat_get_tab_entry(dbentry, msg->m_objectid, false);
+ if (tabentry != NULL)
+ {
+ tabentry->frozen_page_marks_cleared = 0;
+ tabentry->visible_page_marks_cleared = 0;
+ tabentry->total_vacuum_time = 0;
+ tabentry->total_autovacuum_time = 0;
+ tabentry->total_vacuum_delay_time = 0;
+ tabentry->total_autovacuum_delay_time = 0;
+ tabentry->vacuum_failsafe_count = 0;
+ if (pgstat_track_vacuum_statistics)
+ MemSet(&tabentry->vacuum_stats, 0, sizeof(tabentry->vacuum_stats));
+ }
+ return;
+ }
+
+ hash_seq_init(&hstat, dbentry->tables);
+ while ((tabentry = (PgStat_StatTabEntry *) hash_seq_search(&hstat)) != NULL)
+ {
+ tabentry->frozen_page_marks_cleared = 0;
+ tabentry->visible_page_marks_cleared = 0;
+ tabentry->total_vacuum_time = 0;
+ tabentry->total_autovacuum_time = 0;
+ tabentry->total_vacuum_delay_time = 0;
+ tabentry->total_autovacuum_delay_time = 0;
+ tabentry->vacuum_failsafe_count = 0;
+ if (pgstat_track_vacuum_statistics)
+ MemSet(&tabentry->vacuum_stats, 0, sizeof(tabentry->vacuum_stats));
+ }
+ dbentry->n_frozen_page_marks_cleared = 0;
+ dbentry->n_visible_page_marks_cleared = 0;
+ dbentry->total_vacuum_time = 0;
+ dbentry->total_autovacuum_time = 0;
+ dbentry->total_vacuum_delay_time = 0;
+ dbentry->total_autovacuum_delay_time = 0;
+ dbentry->vacuum_failsafe_count = 0;
+ dbentry->vacuum_interrupt_count = 0;
+ if (pgstat_track_vacuum_statistics)
+ MemSet(&dbentry->n_vacuum_stats, 0, sizeof(dbentry->n_vacuum_stats));
+}
+
+
/* ----------
* pgstat_recv_analyze() -
*
@@ -5666,11 +6034,13 @@ pgstat_recv_analyze(PgStat_MsgAnalyze *msg, int len)
{
tabentry->autovac_analyze_timestamp = msg->m_analyzetime;
tabentry->autovac_analyze_count++;
+ tabentry->total_autoanalyze_time += msg->m_elapsedtime;
}
else
{
tabentry->analyze_timestamp = msg->m_analyzetime;
tabentry->analyze_count++;
+ tabentry->total_analyze_time += msg->m_elapsedtime;
}
}
diff --git a/src/backend/tsearch/ts_typanalyze.c b/src/backend/tsearch/ts_typanalyze.c
index 504ba1569ee..5c5ff6d8dc1 100644
--- a/src/backend/tsearch/ts_typanalyze.c
+++ b/src/backend/tsearch/ts_typanalyze.c
@@ -206,7 +206,7 @@ compute_tsvector_stats(VacAttrStats *stats,
char *lexemesptr;
int j;
- vacuum_delay_point();
+ vacuum_delay_point(true);
value = fetchfunc(stats, vector_no, &isnull);
diff --git a/src/backend/utils/adt/array_typanalyze.c b/src/backend/utils/adt/array_typanalyze.c
index 8993d23e18b..71f99bbd327 100644
--- a/src/backend/utils/adt/array_typanalyze.c
+++ b/src/backend/utils/adt/array_typanalyze.c
@@ -314,7 +314,7 @@ compute_array_stats(VacAttrStats *stats, AnalyzeAttrFetchFunc fetchfunc,
int distinct_count;
bool count_item_found;
- vacuum_delay_point();
+ vacuum_delay_point(true);
value = fetchfunc(stats, array_no, &isnull);
if (isnull)
diff --git a/src/backend/utils/adt/rangetypes_typanalyze.c b/src/backend/utils/adt/rangetypes_typanalyze.c
index 9d5cf897c45..444c6c2da8e 100644
--- a/src/backend/utils/adt/rangetypes_typanalyze.c
+++ b/src/backend/utils/adt/rangetypes_typanalyze.c
@@ -168,7 +168,7 @@ compute_range_stats(VacAttrStats *stats, AnalyzeAttrFetchFunc fetchfunc,
upper;
float8 length;
- vacuum_delay_point();
+ vacuum_delay_point(true);
value = fetchfunc(stats, range_no, &isnull);
if (isnull)
diff --git a/src/backend/utils/error/elog.c b/src/backend/utils/error/elog.c
index 6c8db2ef5fa..e5566c413ae 100644
--- a/src/backend/utils/error/elog.c
+++ b/src/backend/utils/error/elog.c
@@ -1647,6 +1647,23 @@ geterrcode(void)
return edata->sqlerrcode;
}
+/*
+ * geterrlevel --- return the elevel of the error currently being constructed
+ *
+ * This is only intended for use in error callback subroutines, where it lets
+ * a callback tell a genuine error apart from a lower-severity report.
+ */
+int
+geterrlevel(void)
+{
+ ErrorData *edata = &errordata[errordata_stack_depth];
+
+ /* we don't bother incrementing recursion_depth */
+ CHECK_STACK_DEPTH();
+
+ return edata->elevel;
+}
+
/*
* geterrposition --- return the currently set error position (0 if none)
*
diff --git a/src/backend/utils/misc/guc.c b/src/backend/utils/misc/guc.c
index 6166c5ff249..ef548d23930 100644
--- a/src/backend/utils/misc/guc.c
+++ b/src/backend/utils/misc/guc.c
@@ -1627,6 +1627,24 @@ static struct config_bool ConfigureNamesBool[] =
true,
NULL, NULL, NULL
},
+ {
+ {"track_vacuum_statistics", PGC_POSTMASTER, STATS_COLLECTOR,
+ gettext_noop("Collects statistics on what vacuum did and what it cost."),
+ gettext_noop("The counters are exposed by the vacuum_stats extension.")
+ },
+ &pgstat_track_vacuum_statistics,
+ false,
+ NULL, NULL, NULL
+ },
+ {
+ {"track_cost_delay_timing", PGC_SUSET, STATS_COLLECTOR,
+ gettext_noop("Collects timing statistics for cost-based vacuum delay."),
+ NULL
+ },
+ &track_cost_delay_timing,
+ false,
+ NULL, NULL, NULL
+ },
{
{"track_io_timing", PGC_SUSET, STATS_COLLECTOR,
gettext_noop("Collects timing statistics for database I/O activity."),
diff --git a/src/backend/utils/misc/postgresql.conf.sample b/src/backend/utils/misc/postgresql.conf.sample
index 4192dfb2748..0487835dd38 100644
--- a/src/backend/utils/misc/postgresql.conf.sample
+++ b/src/backend/utils/misc/postgresql.conf.sample
@@ -626,6 +626,8 @@ optimizer_analyze_root_partition = on # stats collection on root partitions
#track_activities = on
#track_activity_query_size = 1024 # (change requires restart)
#track_counts = off
+#track_vacuum_statistics = off # collect vacuum work and cost (change requires restart)
+#track_cost_delay_timing = off
#track_io_timing = off
#track_wal_io_timing = off
#track_functions = none # none, pl, all
diff --git a/src/include/access/appendonly_compaction.h b/src/include/access/appendonly_compaction.h
index 44aa78a39fb..59a552adfee 100644
--- a/src/include/access/appendonly_compaction.h
+++ b/src/include/access/appendonly_compaction.h
@@ -13,6 +13,7 @@
#ifndef APPENDONLY_COMPACTION_H
#define APPENDONLY_COMPACTION_H
+#include "datatype/timestamp.h"
#include "nodes/pg_list.h"
#include "access/appendonly_visimap.h"
#include "utils/rel.h"
@@ -28,9 +29,17 @@
*/
typedef struct AOVacuumRelStats
{
- int nbytes_truncated; /* current # of bytes truncated from segment file */
- int num_dead_tuples; /* current # of dead tuples */
+ TimestampTz starttime; /* start of the first vacuum phase in this worker */
+ int64 nbytes_truncated; /* current # of bytes truncated from segment file */
+ int64 num_dead_tuples; /* current # of dead tuples */
int num_index_vacuumed; /* current # of indexes been vacuumed */
+
+ /* for the vacuum statistics, accumulated over all the phases */
+ int64 vacuum_time; /* time spent in the phases, in microseconds */
+ int64 phase_start_delay; /* counter at the start of the current phase */
+ int64 delay_time; /* of which the cost-based vacuum delay */
+ int64 dead_tuples_left; /* tuples the post-cleanup found still hidden */
+ int64 total_file_segs; /* segment metadata entries after post-cleanup */
} AOVacuumRelStats;
extern Bitmapset *AppendOptimizedCollectDeadSegments(Relation aorel);
diff --git a/src/include/commands/vacuum.h b/src/include/commands/vacuum.h
index e33d39973c0..7b80970f0cd 100644
--- a/src/include/commands/vacuum.h
+++ b/src/include/commands/vacuum.h
@@ -371,6 +371,8 @@ extern int vacuum_multixact_failsafe_age;
extern pg_atomic_uint32 *VacuumSharedCostBalance;
extern pg_atomic_uint32 *VacuumActiveNWorkers;
extern int VacuumCostBalanceLocal;
+extern PGDLLIMPORT int64 VacuumDelayTime;
+extern PGDLLIMPORT bool track_cost_delay_timing;
/* in commands/vacuum.c */
@@ -409,7 +411,7 @@ extern void vacuum_set_xid_limits(Relation rel,
extern bool vacuum_xid_failsafe_check(TransactionId relfrozenxid,
MultiXactId relminmxid);
extern void vac_update_datfrozenxid(void);
-extern void vacuum_delay_point(void);
+extern void vacuum_delay_point(bool is_analyze);
extern bool vacuum_is_relation_owner(Oid relid, Form_pg_class reltuple,
bits32 options);
extern Relation vacuum_open_relation(Oid relid, RangeVar *relation,
diff --git a/src/include/pgstat.h b/src/include/pgstat.h
index c54b1bb369a..717090a07dc 100644
--- a/src/include/pgstat.h
+++ b/src/include/pgstat.h
@@ -85,6 +85,8 @@ typedef enum StatMsgType
PGSTAT_MTYPE_REPLSLOT,
PGSTAT_MTYPE_CONNECT,
PGSTAT_MTYPE_DISCONNECT,
+ PGSTAT_MTYPE_VACSTATS,
+ PGSTAT_MTYPE_RESETVACSTATS,
} StatMsgType;
/* ----------
@@ -133,6 +135,9 @@ typedef struct PgStat_TableCounts
PgStat_Counter t_blocks_fetched;
PgStat_Counter t_blocks_hit;
+
+ PgStat_Counter t_frozen_page_marks_cleared;
+ PgStat_Counter t_visible_page_marks_cleared;
} PgStat_TableCounts;
/* Possible targets for resetting cluster-wide shared values */
@@ -282,7 +287,7 @@ typedef struct PgStat_TableEntry
* ----------
*/
#define PGSTAT_NUM_TABENTRIES \
- ((PGSTAT_MSG_PAYLOAD - sizeof(Oid) - 3 * sizeof(int) - 5 * sizeof(PgStat_Counter)) \
+ ((PGSTAT_MSG_PAYLOAD - sizeof(Oid) - 3 * sizeof(int) - 6 * sizeof(PgStat_Counter)) \
/ sizeof(PgStat_TableEntry))
typedef struct PgStat_MsgTabstat
@@ -297,6 +302,7 @@ typedef struct PgStat_MsgTabstat
PgStat_Counter m_session_time;
PgStat_Counter m_active_time;
PgStat_Counter m_idle_in_xact_time;
+ PgStat_Counter m_vacuum_interrupt_count;
PgStat_TableEntry m_entry[PGSTAT_NUM_TABENTRIES];
} PgStat_MsgTabstat;
@@ -413,12 +419,55 @@ typedef struct PgStat_MsgVacuum
Oid m_databaseid;
Oid m_tableoid;
bool m_autovacuum;
+ bool m_isindex; /* index time does not update tuple/count fields or DB totals */
TimestampTz m_vacuumtime;
PgStat_Counter m_live_tuples;
PgStat_Counter m_dead_tuples;
+ PgStat_Counter m_elapsedtime; /* microseconds */
+ PgStat_Counter m_delaytime; /* microseconds */
+ bool m_failsafe; /* this completed table vacuum entered failsafe */
} PgStat_MsgVacuum;
+/* ----------
+ * PgStat_VacuumStats Vacuum statistics reported for a relation.
+ * ----------
+ */
+typedef struct PgStat_VacuumStats
+{
+ PgStat_Counter tuples_deleted; /* tuples removed by vacuum */
+ PgStat_Counter dead_tuples; /* dead tuples left unremoved */
+ PgStat_Counter pages_deleted; /* pages removed/deleted by vacuum */
+ PgStat_Counter bytes_removed; /* bytes physically truncated from table files */
+ PgStat_Counter dead_pages; /* pages with unremoved dead tuples */
+ PgStat_Counter total_file_segs; /* latest AO segment count, not cumulative */
+ PgStat_Counter pages_frozen; /* pages where vacuum froze tuples */
+ PgStat_Counter pages_all_visible; /* pages marked all-visible by vacuum */
+
+ /*
+ * Number of heap vacuum runs made aggressive by the XID or MultiXact
+ * freeze table age. This includes VACUUM FREEZE, which sets those
+ * age thresholds to zero, but not DISABLE_PAGE_SKIPPING alone.
+ */
+ PgStat_Counter freeze_age_vacuum_count;
+} PgStat_VacuumStats;
+
+/* ----------
+ * PgStat_MsgVacstats Sent by the backend or autovacuum daemon
+ * after vacuuming a heap relation or an index
+ * to report per-relation vacuum counters.
+ * ----------
+ */
+typedef struct PgStat_MsgVacstats
+{
+ PgStat_MsgHdr m_hdr;
+ Oid m_databaseid;
+ Oid m_tableoid;
+ bool m_isindex; /* counted apart from the database totals */
+ PgStat_VacuumStats m_stats;
+} PgStat_MsgVacstats;
+
+
/* ----------
* PgStat_MsgAnalyze Sent by the backend or autovacuum daemon
* after ANALYZE
@@ -434,6 +483,7 @@ typedef struct PgStat_MsgAnalyze
TimestampTz m_analyzetime;
PgStat_Counter m_live_tuples;
PgStat_Counter m_dead_tuples;
+ PgStat_Counter m_elapsedtime; /* microseconds */
} PgStat_MsgAnalyze;
@@ -685,6 +735,20 @@ typedef struct PgStat_MsgDisconnect
SessionEndType m_cause;
} PgStat_MsgDisconnect;
+/* ----------
+ * PgStat_MsgResetVacstats Sent by the backend to throw away the vacuum
+ * counters of one relation, or of the whole
+ * database when m_resetall is true.
+ * ----------
+ */
+typedef struct PgStat_MsgResetVacstats
+{
+ PgStat_MsgHdr m_hdr;
+ Oid m_databaseid;
+ Oid m_objectid;
+ bool m_resetall;
+} PgStat_MsgResetVacstats;
+
/* ----------
* PgStat_Msg Union over all possible messages.
* ----------
@@ -704,6 +768,8 @@ typedef union PgStat_Msg
PgStat_MsgResetreplslotcounter msg_resetreplslotcounter;
PgStat_MsgAutovacStart msg_autovacuum_start;
PgStat_MsgVacuum msg_vacuum;
+ PgStat_MsgVacstats msg_vacstats;
+ PgStat_MsgResetVacstats msg_resetvacstats;
PgStat_MsgAnalyze msg_analyze;
PgStat_MsgArchiver msg_archiver;
PgStat_MsgQueuestat msg_queuestat; /* GPDB */
@@ -730,7 +796,7 @@ typedef union PgStat_Msg
* ------------------------------------------------------------
*/
-#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA2
+#define PGSTAT_FILE_FORMAT_ID 0x01A5BCAD
/* ----------
* PgStat_StatDBEntry The collector's data per database
@@ -769,15 +835,32 @@ typedef struct PgStat_StatDBEntry
PgStat_Counter n_sessions_fatal;
PgStat_Counter n_sessions_killed;
+ /* Cumulative table vacuum times; index work is already included. */
+ PgStat_Counter total_vacuum_time; /* microseconds */
+ PgStat_Counter total_autovacuum_time; /* microseconds */
+ PgStat_Counter total_vacuum_delay_time; /* microseconds */
+ PgStat_Counter total_autovacuum_delay_time; /* microseconds */
+ PgStat_Counter vacuum_failsafe_count;
+
+ /* Heap vacuums in this database interrupted by ERROR. */
+ PgStat_Counter vacuum_interrupt_count;
+
+ /* VM revisions are fed by ordinary relation statistics. */
+ PgStat_Counter n_frozen_page_marks_cleared;
+ PgStat_Counter n_visible_page_marks_cleared;
+
TimestampTz stat_reset_timestamp;
TimestampTz stats_timestamp; /* time of db stats file update */
/*
- * tables and functions must be last in the struct, because we don't write
- * the pointers out to the stats file.
+ * Only the prefix before these pointers is written as ordinary database
+ * statistics. The optional vacuum counters are serialized separately.
*/
HTAB *tables;
HTAB *functions;
+
+ /* Must be last: storage is omitted when tracking is disabled at startup. */
+ PgStat_VacuumStats n_vacuum_stats;
} PgStat_StatDBEntry;
@@ -816,8 +899,32 @@ typedef struct PgStat_StatTabEntry
PgStat_Counter analyze_count;
TimestampTz autovac_analyze_timestamp; /* autovacuum initiated */
PgStat_Counter autovac_analyze_count;
+
+ /* Cumulative maintenance times, in microseconds. */
+ PgStat_Counter total_vacuum_time;
+ PgStat_Counter total_autovacuum_time;
+ PgStat_Counter total_analyze_time;
+ PgStat_Counter total_autoanalyze_time;
+ PgStat_Counter total_vacuum_delay_time;
+ PgStat_Counter total_autovacuum_delay_time;
+ PgStat_Counter vacuum_failsafe_count;
+
+ /* VM revisions are fed by ordinary relation statistics. */
+ PgStat_Counter frozen_page_marks_cleared;
+ PgStat_Counter visible_page_marks_cleared;
+
+ /* Must be last: storage is omitted when tracking is disabled at startup. */
+ PgStat_VacuumStats vacuum_stats;
} PgStat_StatTabEntry;
+/* The postmaster setting fixes hash entry sizes for the process lifetime. */
+#define PGSTAT_DB_ENTRY_SIZE \
+ (pgstat_track_vacuum_statistics ? sizeof(PgStat_StatDBEntry) : \
+ offsetof(PgStat_StatDBEntry, n_vacuum_stats))
+#define PGSTAT_TAB_ENTRY_SIZE \
+ (pgstat_track_vacuum_statistics ? sizeof(PgStat_StatTabEntry) : \
+ offsetof(PgStat_StatTabEntry, vacuum_stats))
+
/* ----------
* PgStat_StatQueueEntry The collector's data per resource queue
@@ -986,6 +1093,7 @@ typedef struct PgStat_FunctionCallUsage
* ----------
*/
extern PGDLLIMPORT bool pgstat_track_counts;
+extern PGDLLIMPORT bool pgstat_track_vacuum_statistics;
extern PGDLLIMPORT int pgstat_track_functions;
extern char *pgstat_stat_directory;
extern char *pgstat_stat_tmpname;
@@ -1059,10 +1167,32 @@ extern void pgstat_reset_replslot_counter(const char *name);
extern void pgstat_report_connect(Oid dboid);
extern void pgstat_report_autovac(Oid dboid);
extern void pgstat_report_vacuum(Oid tableoid, bool shared,
- PgStat_Counter livetuples, PgStat_Counter deadtuples);
+ PgStat_Counter livetuples, PgStat_Counter deadtuples,
+ TimestampTz starttime, PgStat_Counter delaytime,
+ bool failsafe);
+extern void pgstat_count_vacuum_error(bool shared);
+extern void pgstat_report_index_vacuum_time(Relation rel,
+ PgStat_Counter elapsedtime,
+ PgStat_Counter delaytime, bool is_autovacuum);
+extern void pgstat_report_vacstats(Oid tableoid, bool shared, bool isindex,
+ const PgStat_VacuumStats *stats);
+extern void pgstat_reset_vacuum_stats(Oid relid, bool resetall);
+
+/* count a page whose all-visible bit is being cleared */
+#define pgstat_count_visible_page_marks_cleared(rel) \
+ do { \
+ if ((rel)->pgstat_info != NULL) \
+ (rel)->pgstat_info->t_counts.t_visible_page_marks_cleared++; \
+ } while (0)
+/* count a page whose all-frozen bit is being cleared */
+#define pgstat_count_frozen_page_marks_cleared(rel) \
+ do { \
+ if ((rel)->pgstat_info != NULL) \
+ (rel)->pgstat_info->t_counts.t_frozen_page_marks_cleared++; \
+ } while (0)
extern void pgstat_report_analyze(Relation rel,
PgStat_Counter livetuples, PgStat_Counter deadtuples,
- bool resetcounter);
+ bool resetcounter, TimestampTz starttime);
extern void pgstat_report_recovery_conflict(int reason);
extern void pgstat_report_deadlock(void);
@@ -1270,6 +1400,7 @@ extern void pgstat_combine_from_qe(struct CdbDispatchResults *results, /* GPDB *
*/
extern PgStat_StatDBEntry *pgstat_fetch_stat_dbentry(Oid dbid);
extern PgStat_StatTabEntry *pgstat_fetch_stat_tabentry(Oid relid);
+extern PgStat_VacuumStats *pgstat_fetch_stat_vacuum_stats(Oid relid);
extern PgStat_StatQueueEntry *pgstat_fetch_stat_queueentry(Oid queueid); /* GPDB */
extern PgBackendStatus *pgstat_fetch_stat_beentry(int beid);
diff --git a/src/include/utils/elog.h b/src/include/utils/elog.h
index 001bc08a3a5..97f218e907c 100644
--- a/src/include/utils/elog.h
+++ b/src/include/utils/elog.h
@@ -263,6 +263,7 @@ extern void internalerrquery(const char *query);
extern void err_generic_string(int field, const char *str);
extern int geterrcode(void);
+extern int geterrlevel(void);
extern int geterrposition(void);
extern int getinternalerrposition(void);
diff --git a/src/include/utils/unsync_guc_name.h b/src/include/utils/unsync_guc_name.h
index 85ecb3548e6..9b98358a478 100644
--- a/src/include/utils/unsync_guc_name.h
+++ b/src/include/utils/unsync_guc_name.h
@@ -599,9 +599,11 @@
"track_activities",
"track_activity_query_size",
"track_commit_timestamp",
+ "track_cost_delay_timing",
"track_counts",
"track_functions",
"track_io_timing",
+ "track_vacuum_statistics",
"transaction_deferrable",
"transaction_isolation",
"transaction_read_only",