diff --git a/contrib/Makefile b/contrib/Makefile index 5ea76366363..5d8e38819d6 100644 --- a/contrib/Makefile +++ b/contrib/Makefile @@ -54,6 +54,7 @@ SUBDIRS = \ tsm_system_rows \ tsm_system_time \ unaccent \ + vacuum_stats \ vacuumlo # Cloudberry-specific additions (to ease merge pain). diff --git a/contrib/bloom/blvacuum.c b/contrib/bloom/blvacuum.c index 88b0a6d2900..d8873f96822 100644 --- a/contrib/bloom/blvacuum.c +++ b/contrib/bloom/blvacuum.c @@ -61,7 +61,7 @@ blbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats, *itupPtr, *itupEnd; - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, info->strategy); @@ -191,7 +191,7 @@ blvacuumcleanup(IndexVacuumInfo *info, IndexBulkDeleteResult *stats) Buffer buffer; Page page; - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, info->strategy); diff --git a/contrib/file_fdw/file_fdw.c b/contrib/file_fdw/file_fdw.c index cce94a5b335..5e870115661 100644 --- a/contrib/file_fdw/file_fdw.c +++ b/contrib/file_fdw/file_fdw.c @@ -1170,7 +1170,7 @@ file_acquire_sample_rows(Relation onerel, int elevel, for (;;) { /* Check for user-requested abort or sleep */ - vacuum_delay_point(); + vacuum_delay_point(true); /* Fetch next row */ MemoryContextReset(tupcontext); diff --git a/contrib/vacuum_stats/.gitignore b/contrib/vacuum_stats/.gitignore new file mode 100644 index 00000000000..5dcb3ff9723 --- /dev/null +++ b/contrib/vacuum_stats/.gitignore @@ -0,0 +1,4 @@ +# Generated subdirectories +/log/ +/results/ +/tmp_check/ diff --git a/contrib/vacuum_stats/Makefile b/contrib/vacuum_stats/Makefile new file mode 100644 index 00000000000..aee5f608665 --- /dev/null +++ b/contrib/vacuum_stats/Makefile @@ -0,0 +1,29 @@ +# contrib/vacuum_stats/Makefile + +MODULE_big = vacuum_stats +OBJS = vacuum_stats.o + +EXTENSION = vacuum_stats +DATA = vacuum_stats--1.0.sql +PGFILEDESC = "vacuum_stats - per-relation and per-database vacuum statistics" + +TAP_TESTS = 1 + +ifdef USE_PGXS +PG_CONFIG = pg_config +PGXS := $(shell $(PG_CONFIG) --pgxs) +include $(PGXS) +else +subdir = contrib/vacuum_stats +top_builddir = ../.. +include $(top_builddir)/src/Makefile.global +include $(top_srcdir)/contrib/contrib-global.mk +endif + +check-tap: + $(prove_check) + +installcheck-tap: + $(prove_installcheck) + +.PHONY: check-tap installcheck-tap diff --git a/contrib/vacuum_stats/README.md b/contrib/vacuum_stats/README.md new file mode 100644 index 00000000000..4f1252dc2ec --- /dev/null +++ b/contrib/vacuum_stats/README.md @@ -0,0 +1,139 @@ +# vacuum_stats + +`vacuum_stats` describes the work done by VACUUM for tables, indexes and +whole databases: how many tuples it removes, what remains to be cleaned, +how it changes page visibility, and how much time it spends. These counters +help evaluate the results and cost of vacuuming over time, alongside the +existing vacuum counts and timestamps. + +Collection of extended work counters is enabled with `track_vacuum_statistics = on` in the server +configuration and requires a restart. Set it consistently on the coordinator +and segments. When enabled, the extended vacuum counters and database totals +are stored with ordinary relation and database statistics. When disabled, their storage +is omitted and those extended fields return zero. Restarting with tracking +disabled discards the extended counters while preserving ordinary statistics, +including vacuum times, VM clearings, failsafe and interruption counts. + +The `visible_page_marks_cleared` and `frozen_page_marks_cleared` counters describe +visibility-map changes caused by data modifications. They follow `track_counts` +and are collected and retained independently of `track_vacuum_statistics`. +The dedicated vacuum-statistics reset functions also reset these counters, +vacuum times and failsafe counts. Database-wide vacuum reset also +clears `pg_stat_vacuum_database.vacuum_interrupt_count`; relation reset preserves +this database total. The reset functions preserve ordinary access counters, +maintenance counts and timestamps, and ANALYZE times. + +Install with `CREATE EXTENSION vacuum_stats`. All new SQL functions and +views belong to the extension's schema. Existing system views and built-in +function OIDs remain unchanged. VM clearings, maintenance times, failsafe and +interruption counts follow `track_counts` independently of +`track_vacuum_statistics`; the extension reads their ordinary collector entries. + +Cost-based delay timing additionally requires `track_cost_delay_timing = on`. +It is disabled by default and can be changed for a session without restarting. +ANALYZE elapsed times are recorded separately from VACUUM times. ANALYZE +sampling delays do not contribute to the VACUUM delay counters. + +The counters accumulate until reset, except `total_file_segs`, which records +the state observed by the last completed AO VACUUM. To examine a particular period, compare +two readings without an intervening reset or change in tracking configuration. + +## What the counters measure + +| Counter | Meaning | +|---------|---------| +| `tuples_deleted` | Table tuples or index entries removed by VACUUM. | +| `dead_tuples` | Heap tuples found dead but not yet removable; for AO tables, hidden tuples remaining after vacuum. | +| `pages_deleted` | Pages truncated from a heap, newly deleted index pages, or space freed from AO segment files expressed in blocks. | +| `bytes_removed` | Bytes physically truncated from heap or AO table files, including AO tails left by aborted inserts. Index page reuse does not increase this counter. | +| `dead_pages` | Heap pages containing unremovable dead tuples; for indexes, deleted pages not yet available for reuse. | +| `pages_frozen` | Heap pages on which VACUUM froze at least one tuple. | +| `pages_all_visible` | Heap pages VACUUM marked all-visible. | +| `visible_page_marks_cleared` | Clearings of the all-visible flag, usually caused by data changes. | +| `frozen_page_marks_cleared` | Clearings of the all-frozen flag, usually caused by data changes. | +| `freeze_age_vacuum_count` | Heap vacuum runs made aggressive by transaction or multixact freeze age. This includes `VACUUM FREEZE`; forcing page scanning alone does not increment it. | +| `vacuum_failsafe_count` | Completed heap vacuum runs that entered failsafe mode to avoid transaction or multixact wraparound. Aggressive scanning alone does not increment it. | +| `total_vacuum_time`, `total_autovacuum_time` | Elapsed time in milliseconds, separately for manual VACUUM and autovacuum. | +| `total_vacuum_delay_time`, `total_autovacuum_delay_time` | Cost-based delays in milliseconds, included in the corresponding elapsed time. | +| `total_analyze_time`, `total_autoanalyze_time` | Table ANALYZE time in milliseconds, separately for manual and automatic runs. | +| `vacuum_interrupt_count` | Database count of heap vacuums interrupted by an ERROR, including cancellation, while the vacuum error callback is installed. | + +`total_file_segs` in the table views records the number of AO segment metadata +entries observed after the last VACUUM, including empty and awaiting-drop +segments. For AO column tables it counts logical segments, not each column +file. A subsequent VACUUM replaces this value instead of adding to it. It is +zero before the first report, after reset, and for heap tables. Use it with +`bytes_removed` and remaining hidden tuples to understand compaction results. + +The other fields are cumulative counters, not a snapshot of the table's current +contents. In particular, successive runs can count the same unremovable +tuple or page again. Visibility flags can also be set and cleared repeatedly. +A counter difference measures work or observations during the interval, +not necessarily a number of distinct tuples or pages. + +## How to use them + +- **Find expensive relations.** Compare increases in `total_vacuum_time` and + `total_autovacuum_time` between tables and indexes over the same interval. Relate that time to tuples + removed and pages reclaimed to see where maintenance time is spent. + Low tuple removal alone does not imply wasted work: vacuum also freezes + tuples and maintains visibility information. +- **Find work that cannot finish.** Repeated increases in heap `dead_tuples` + and `dead_pages` show that vacuum keeps encountering data it cannot remove. + Check for old snapshots or long-running transactions before increasing + vacuum frequency. +- **Separate throttling from other costs.** Compare `total_vacuum_delay_time` with + `total_vacuum_time`, and the corresponding autovacuum counters. A large share + spent in cost-based delays helps explain a long run. The remaining time includes execution and other waits; it is + not a measurement of CPU time. +- **Understand visibility and freezing work.** Compare `pages_all_visible` + with `visible_page_marks_cleared` to see how often data changes undo visibility + work. Use `pages_frozen`, `frozen_page_marks_cleared` and + `freeze_age_vacuum_count` to understand freezing activity and aggressive + scans. Zero frozen pages can be normal when no tuples need freezing. +- **Compare segments.** Differences in work and time for the same relation + can help identify uneven data distribution or different execution costs. + Compare both quantities: a slower segment is not necessarily processing + more data. Summed segment time represents accumulated work, not the + wall-clock duration of a distributed VACUUM. + +## Tables, indexes and AO + +A table's vacuum time includes its index maintenance. Database totals +include table work without adding the index counters again. Use the index +figures to understand that part of the cost, rather than adding them to +table totals. Deleted index pages can become reusable within the index; +they do not necessarily represent space returned to the filesystem. + +For AO row and column tables, `tuples_deleted` measures rows discarded by +compaction, and `dead_tuples` counts hidden rows left afterwards. Hidden +rows can remain when compaction is disabled or a segment file is not +eligible for compaction. AO index cleanup can remove entries for relocated +live rows as well as deleted rows, so its tuple count can exceed the table's. +AO `pages_deleted` expresses truncated bytes in blocks, rounded up; +`bytes_removed` preserves the exact byte count. Freed space can include live +rows moved to new segment files, so this is not the net reduction in table size. +AO elapsed time covers the interval from the first phase seen by the +reporting worker to final cleanup, including gaps between those phases. + +Heap page visibility, freezing and failsafe counters do not apply to the AO table +itself. Its auxiliary heap relations have their own statistics. Indexes +have tuple-removal, page-deletion and timing counters, but no heap visibility +or freezing work. + +The `pg_stat_vacuum_tables`, `pg_stat_vacuum_indexes` and +`pg_stat_vacuum_database` views expose local statistics. Their +`gp_stat_vacuum_*` counterparts include the coordinator and segments, +identified by `gp_segment_id`. In utility mode there is no dispatch: these +views return the connected node's local statistics once, with its segment ID. + +The database views include `datid = 0`, with a null `datname`, for shared +relations. `vacuum_interrupt_count` does not include VACUUM FULL, failures +before the heap callback is installed, or AO parent compaction. Auxiliary +heap vacuums are counted independently. An interrupted run does not report +its usual completion counters, so check errors when successful-run totals +alone do not explain maintenance activity. + +The on-disk statistics format changes, so older saved statistics are discarded +on first start. The system catalog version is unchanged; these statistics do +not require a new cluster or replacement of system views. diff --git a/contrib/vacuum_stats/t/001_vacuum_statistics.pl b/contrib/vacuum_stats/t/001_vacuum_statistics.pl new file mode 100644 index 00000000000..8fc4d92558c --- /dev/null +++ b/contrib/vacuum_stats/t/001_vacuum_statistics.pl @@ -0,0 +1,822 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Test vacuum counter semantics, controls and lifecycle in one TAP suite. +# Adapt the repeatable-read and visibility-map scenarios from v44: +# https://www.postgresql.org/message-id/flat/cb305107-5935-4c34-9847-6ff0fef89f06%40yandex.ru#8b78d4ab399e37afc5eb2cee8a54d2f6 +# This branch has a UDP collector and the older pg_stat_vacuum_* API: +# dead_tuples corresponds to recently_dead_tuples, and the page counters +# expose freezing and VM transitions. Polling uses a fresh connection, so +# it does not reuse a cached statistics snapshot. Core VM clearing semantics +# are covered by 004_visibility_map_stats.pl; here check vacuum +# reports, the extension views, controls and lifecycle. +# PostgresNode runs in utility/maintenance mode; use the local views. +# Read ordinary counters from pg_stat_all_tables_internal: the public +# pg_stat_all_tables gathers segment data and has no user-table rows here. + +use strict; +use warnings; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('vacuum_stats'); +$node->init; +$node->append_conf('postgresql.conf', q{ +autovacuum = off +track_counts = on +vacuum_cost_delay = 0 +}); +$node->start; +$node->safe_psql('postgres', q{ +CREATE EXTENSION vacuum_stats; +CREATE TABLE vstat_barrier (id int) WITH (autovacuum_enabled = off); +}); + +my @counter_names = qw(tuples_deleted dead_tuples pages_deleted bytes_removed dead_pages + pages_frozen pages_all_visible frozen_page_marks_cleared visible_page_marks_cleared + freeze_age_vacuum_count vacuum_failsafe_count total_vacuum_time total_autovacuum_time + total_vacuum_delay_time total_autovacuum_delay_time); +my $all_counters = join(', ', @counter_names); +my $all_zero = join(' AND ', map { "$_ = 0" } @counter_names); +my $vacuum_zero = join(' AND ', map { "$_ = 0" } grep { !/_page_marks_cleared$|^total_.*time$|^vacuum_failsafe_count$/ } @counter_names); +my $vm_columns = 'frozen_page_marks_cleared, visible_page_marks_cleared'; + +sub wait_for_stats +{ + my ($sql, $description, $expected) = @_; + $expected = 't' unless defined $expected; + $node->poll_query_until('postgres', $sql, $expected) + or BAIL_OUT("timed out waiting for $description: $sql"); +} + +# VACUUM sends vacuum_count before its extended counters. Instead of +# using that earlier report, send ANALYZE from the same backend AFTER the +# command and wait for its report. ANALYZE adds no vacuum counters, so it +# also works for database totals, disabled tracking, resets and VACUUM FULL. +# DML counters are buffered separately; wait for the affected table's DML +# report explicitly when checking VM clearing counters. +sub run_and_wait +{ + my ($sql) = @_; + my $count = $node->safe_psql('postgres', + "SELECT analyze_count FROM pg_stat_all_tables_internal WHERE relname = 'vstat_barrier'"); + BAIL_OUT("missing local ANALYZE count for vstat_barrier") + unless $count =~ /^\d+$/; + my ($result, $stdout, $stderr) = $node->psql('postgres', + "$sql;\nANALYZE vstat_barrier;", on_error_die => 1); + wait_for_stats( + "SELECT analyze_count = $count + 1 FROM pg_stat_all_tables_internal WHERE relname = 'vstat_barrier'", + "collector report after $sql"); + return $stderr; +} + +sub wait_for_updates +{ + my ($table, $count) = @_; + wait_for_stats( + "SELECT n_tup_upd = $count FROM pg_stat_all_tables_internal WHERE relname = '$table'", + "UPDATE report for $table"); +} + +sub populated_pages +{ + my ($table, $predicate) = @_; + $predicate ||= 'true'; + return $node->safe_psql('postgres', + "SELECT count(DISTINCT split_part(ctid::text, ',', 1)) FROM $table WHERE $predicate"); +} + +sub vacuum_table +{ + my ($table, $options, $settings) = @_; + $options ||= ''; + $settings ||= ''; + return run_and_wait("$settings VACUUM $options $table"); +} + +sub counters +{ + my ($table, $columns) = @_; + $columns ||= $all_counters; + return $node->safe_psql('postgres', + "SELECT $columns FROM pg_stat_vacuum_tables WHERE relname = '$table'"); +} + +sub index_counters +{ + my ($index, $columns) = @_; + $columns ||= $all_counters; + return $node->safe_psql('postgres', + "SELECT $columns FROM pg_stat_vacuum_indexes WHERE indexrelname = '$index'"); +} + +sub database_counters +{ + my ($columns) = @_; + $columns ||= $all_counters; + return $node->safe_psql('postgres', + "SELECT $columns FROM pg_stat_vacuum_database WHERE datname = 'postgres'"); +} + +# Read the same statistics snapshot before measuring its hash allocations. +sub snapshot_bytes +{ + return $node->safe_psql('postgres', q{ +BEGIN; +DO $$ BEGIN PERFORM count(*) FROM pg_stat_all_tables_internal WHERE n_tup_ins >= 0; END $$; +SELECT sum(used_bytes) FROM pg_backend_memory_contexts +WHERE name IN ('Databases hash', 'Per-database table'); +COMMIT; +}); +} + +subtest 'tracking disabled and enabled' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_off (id int) WITH (autovacuum_enabled = off); +INSERT INTO vstat_off SELECT generate_series(1, 1000); +DELETE FROM vstat_off; +}); + my $verbose = vacuum_table('vstat_off', 'VERBOSE'); + like($verbose, qr/table "vstat_off": vacuum statistics.*?cost-based delay: 0\.000 ms/s, + 'VERBOSE reports measurements even with statistics tracking disabled'); + is(counters('vstat_off', $vacuum_zero), 't', + 'extended counters stay zero while tracking is disabled'); + is(counters('vstat_off', 'total_vacuum_time > 0, total_autovacuum_time = 0'), + 't|t', 'core timing remains enabled independently of extended counters'); + is($node->safe_psql('postgres', + "SELECT vacuum_count FROM pg_stat_all_tables_internal WHERE relname = 'vstat_off'"), + '1', 'ordinary vacuum statistics are still collected'); + for my $am ('ao_row', 'ao_column') + { + my $table = "vstat_off_$am"; + $node->safe_psql('postgres', qq{ +CREATE TABLE $table (id int) USING $am; +CREATE INDEX ${table}_idx ON $table (id); +INSERT INTO $table SELECT generate_series(1, 10000); +DELETE FROM $table WHERE id % 2 = 0; +}); + my $verbose = vacuum_table($table, 'VERBOSE', q{ +SET track_cost_delay_timing = on; +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 1; +}); + is(counters($table, "$vacuum_zero AND total_file_segs = 0"), 't', + "$am extended counters and segment snapshot stay zero with tracking off"); + is(index_counters("${table}_idx", $vacuum_zero), 't', + "$am index work counters stay zero with tracking off"); + is(counters($table, 'total_vacuum_time > 0 AND total_vacuum_delay_time > 0'), + 't', "$am table timing remains independent of extended tracking"); + is(index_counters("${table}_idx", 'total_vacuum_time > 0 AND total_vacuum_delay_time > 0'), + 't', "$am compaction index timing remains independent of extended tracking"); + like($verbose, qr/[1-9]\d* bytes truncated; \d+ file segments remain\./, + "$am VERBOSE retains byte and segment measurements with tracking off"); + + # The following run has no obsolete segments and uses scan_index(). + my $index_time = index_counters("${table}_idx", 'total_vacuum_time'); + vacuum_table($table); + is(index_counters("${table}_idx", "total_vacuum_time > $index_time"), 't', + "$am cleanup-only index timing advances with extended tracking off"); + is(index_counters("${table}_idx", $vacuum_zero), 't', + "$am cleanup-only index work counters remain disabled"); + } + is(database_counters($vacuum_zero), 't', + 'heap and AO work do not populate extended database totals with tracking off'); + # VM changes are ordinary DML statistics even with vacuum tracking off. + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_vm_off (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_vm_off SELECT generate_series(1, 1000); +}); + vacuum_table('vstat_vm_off', 'FREEZE'); + my $pages = $node->safe_psql('postgres', + "SELECT pg_relation_size('vstat_vm_off') / current_setting('block_size')::bigint"); + my ($db_frozen, $db_visible) = split /\|/, database_counters($vm_columns); + $node->safe_psql('postgres', 'UPDATE vstat_vm_off SET id = id + 1000'); + wait_for_updates('vstat_vm_off', 1000); + is(counters('vstat_vm_off', $vm_columns), "$pages|$pages", + 'DML counts exact VM clearings while vacuum tracking is off'); + is(database_counters("frozen_page_marks_cleared - $db_frozen, visible_page_marks_cleared - $db_visible"), + "$pages|$pages", 'database VM totals receive the same disabled-tracking DML report'); + is(counters('vstat_vm_off', $vacuum_zero), 't', + 'VM collection does not enable vacuum counters'); + $node->restart; + is(counters('vstat_vm_off', $vm_columns), "$pages|$pages", + 'VM counters survive a clean restart with tracking off'); + # dynahash allocates entries in batches. With only a few relations, + # a larger entry can use a slightly smaller batch and appear cheaper. + # Populate enough ordinary entries to exceed that allocation rounding; + # none of these relations has been vacuumed. + $node->safe_psql('postgres', q{ +CREATE SCHEMA vstat_memory; +DO $$ BEGIN + FOR i IN 1..1024 LOOP + EXECUTE format('CREATE TABLE vstat_memory.t%s (id int) WITH (autovacuum_enabled = off)', i); + EXECUTE format('INSERT INTO vstat_memory.t%s VALUES (1)', i); + END LOOP; +END $$; +}); + wait_for_stats(q{ +SELECT count(*) = 1024 AND bool_and(n_tup_ins = 1) +FROM pg_stat_all_tables_internal WHERE schemaname = 'vstat_memory' +}, 'ordinary statistics for the memory-allocation test'); + my $off_bytes = snapshot_bytes(); + is($node->safe_psql('postgres', + "SELECT context FROM pg_settings WHERE name = 'track_vacuum_statistics'"), + 'postmaster', 'tracking is fixed at server startup'); + my ($result, $stdout, $stderr) = $node->psql('postgres', + 'SET track_vacuum_statistics = on'); + is($result, 3, 'a session cannot enable tracking'); + like($stderr, qr/cannot be changed without restarting the server/, + 'the error explains that a restart is required'); + $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = on'); + $node->reload; + wait_for_stats(q{SELECT pending_restart FROM pg_settings WHERE name = 'track_vacuum_statistics'}, + 'configuration reload to notice the startup setting'); + is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'off', + 'reload leaves tracking disabled'); + $node->restart; + is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'on', + 'restart enables tracking'); + is(counters('vstat_vm_off', $vm_columns), "$pages|$pages", + 'enabling vacuum tracking preserves VM counters collected while off'); + is(counters('vstat_off', $vacuum_zero), 't', + 'new vacuum blocks start at zero when reading ordinary-only statistics'); + cmp_ok(snapshot_bytes(), '>', $off_bytes, + 'enabled snapshots allocate vacuum counters with ordinary relation and DB entries'); + +}; + +subtest 'removed tuples and truncated pages' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_heap (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_heap SELECT generate_series(1, 10000); +DELETE FROM vstat_heap WHERE id % 2 = 0; +}); + is(counters('vstat_heap', $all_zero), 't', + 'all counters are zero before the first vacuum'); + # Heap insertion can pre-extend the relation with empty tail pages. + # Disable truncation in this run to make its zero counter deterministic. + my $verbose = vacuum_table('vstat_heap', '(VERBOSE, TRUNCATE false)'); + is(counters('vstat_heap', 'tuples_deleted, dead_tuples, pages_deleted, dead_pages'), + '5000|0|0|0', 'vacuum removes exactly half the rows without truncation'); + like($verbose, qr/pages with dead tuples not yet removable: 0\n/, + 'VERBOSE reports zero pages retaining dead tuples'); + like($verbose, qr/pages with tuples frozen: 0\n/, + 'VERBOSE reports zero pages frozen with the default freeze age'); + my $visible = counters('vstat_heap', 'pages_all_visible'); + like($verbose, qr/pages marked all-visible: \Q$visible\E\n/, + 'VERBOSE all-visible count matches the first vacuum report'); + like($verbose, qr/scanned index "vstat_heap_pkey".*?cost-based delay: 0\.000 ms/s, + 'VERBOSE reports the index cost delay'); + is(index_counters('vstat_heap_pkey', 'tuples_deleted, pages_deleted'), + '5000|0', 'every other index key remains; no index pages are deleted'); + is(counters('vstat_heap', 'total_vacuum_time > 0, total_autovacuum_time = 0, total_vacuum_delay_time = 0'), + 't|t|t', 'heap vacuum takes time but has no cost delay'); + is(index_counters('vstat_heap_pkey', 'total_vacuum_time > 0, total_autovacuum_time = 0, total_vacuum_delay_time = 0'), + 't|t|t', 'index vacuum takes time but has no cost delay'); + + $node->safe_psql('postgres', 'DELETE FROM vstat_heap'); + my $pages_before = $node->safe_psql('postgres', + "SELECT pg_relation_size('vstat_heap') / current_setting('block_size')::bigint"); + my $quiet = vacuum_table('vstat_heap'); + unlike($quiet, qr/vacuum statistics|pages marked all-visible|cost-based delay/, + 'ordinary VACUUM does not emit VERBOSE statistics at the default message level'); + is(counters('vstat_heap', 'tuples_deleted, dead_tuples, pages_deleted'), + "10000|0|$pages_before", 'the second vacuum removes the remaining rows and truncates every heap page'); + is(counters('vstat_heap', "bytes_removed = pages_deleted * current_setting('block_size')::bigint"), + 't', 'heap truncation reports exact bytes'); + is(index_counters('vstat_heap_pkey', 'tuples_deleted, pages_deleted > 0'), + '10000|t', 'index tuple and page deletion counters accumulate'); +}; + +subtest 'cluster views return local rows once in utility mode' => sub { + is($node->safe_psql('postgres', 'SHOW gp_role'), 'utility', + 'this scenario uses a direct utility connection'); + for my $view ( + ['tables', 'relid, schemaname, relname', "relname = 'vstat_heap'"], + ['indexes', 'relid, indexrelid, schemaname, relname, indexrelname', + "relname = 'vstat_heap'"], + ['database', 'datid, datname', "datname = 'postgres'"]) + { + my ($suffix, $keys, $filter) = @$view; + my $columns = "$keys, $all_counters"; + $columns .= ', total_analyze_time, total_autoanalyze_time' if $suffix eq 'tables'; + $columns .= ', total_file_segs' if $suffix eq 'tables'; + $columns .= ', vacuum_interrupt_count' if $suffix eq 'database'; + my $local = "SELECT $columns FROM pg_stat_vacuum_$suffix WHERE $filter"; + my $cluster = "SELECT $columns FROM gp_stat_vacuum_$suffix WHERE $filter"; + is($node->safe_psql('postgres', qq{ +SELECT count(*) = 1 AND bool_and(gp_segment_id = gp_execution_segment()) +FROM gp_stat_vacuum_$suffix WHERE $filter +}), 't', "$suffix view returns one row identified by the connected node"); + is($node->safe_psql('postgres', qq{ +SELECT NOT EXISTS ( + ($cluster EXCEPT ALL $local) + UNION ALL + ($local EXCEPT ALL $cluster) +) +}), 't', "$suffix cluster and local views have identical rows and counters"); + } +}; + +subtest 'a repeatable-read snapshot prevents removal' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_snapshot (id int PRIMARY KEY, val int) + WITH (autovacuum_enabled = off); +INSERT INTO vstat_snapshot SELECT i, i FROM generate_series(1, 1000) g(i); +}); + my $dead_pages = populated_pages('vstat_snapshot', 'id > 900'); + my ($in, $out) = ('', ''); + my $timer = IPC::Run::timeout($TestLib::timeout_default); + my $reader = $node->background_psql('postgres', \$in, \$out, $timer); + $out = ''; + $in = "BEGIN ISOLATION LEVEL REPEATABLE READ;\n" + . "SELECT count(*) FROM vstat_snapshot;\n\\echo snapshot_ready\n"; + pump_until($reader, $timer, \$out, qr/^snapshot_ready\r?$/m) + or BAIL_OUT('reader did not acquire its snapshot'); + like($out, qr/^1000\r?$/m, 'reader sees all original rows'); + + # Updating the indexed column prevents HOT, making index removals exact. + $node->safe_psql('postgres', 'UPDATE vstat_snapshot SET id = id + 1000 WHERE id > 900'); + my $verbose = vacuum_table('vstat_snapshot', 'VERBOSE'); + is(counters('vstat_snapshot', 'tuples_deleted, dead_tuples, dead_pages, pages_frozen'), + "0|100|$dead_pages|0", '100 old tuple versions remain on exactly the affected pages'); + like($verbose, qr/pages with dead tuples not yet removable: \Q$dead_pages\E\n/, + 'VERBOSE reports the pages held back by the snapshot'); + is(index_counters('vstat_snapshot_pkey', 'tuples_deleted'), + '0', 'index entries needed by the reader are retained'); + + $in = "COMMIT;\n\\q\n"; + $reader->finish; + vacuum_table('vstat_snapshot'); + is(counters('vstat_snapshot', 'tuples_deleted, dead_tuples, pages_frozen'), + '100|100|0', 'after commit, 100 versions are removed; cumulative dead_tuples stays at 100'); + is(index_counters('vstat_snapshot_pkey', 'tuples_deleted'), '100', + 'after commit, the index removes exactly the 100 obsolete entries'); +}; + +subtest 'heap page reports and core VM counters in extension views' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_vm (id int PRIMARY KEY, val int) + WITH (autovacuum_enabled = off, fillfactor = 50); +INSERT INTO vstat_vm SELECT i, i FROM generate_series(1, 5000) g(i); +}); + my $pages = populated_pages('vstat_vm'); + my $verbose = vacuum_table('vstat_vm', 'VERBOSE', + 'SET vacuum_freeze_min_age = 1000000000; SET vacuum_freeze_table_age = 1000000000;'); + is(counters('vstat_vm', 'pages_frozen, pages_all_visible, frozen_page_marks_cleared, visible_page_marks_cleared'), + "0|$pages|0|0", 'ordinary vacuum marks each populated page visible without freezing it'); + like($verbose, qr/pages with tuples frozen: 0\n/, + 'VERBOSE reports no freezing with a high freeze age'); + like($verbose, qr/pages marked all-visible: \Q$pages\E\n/, + 'VERBOSE reports the exact number of newly visible pages'); + + # Leave enough room for new versions on the same pages and prevent HOT. + # The distinct counter values detect swapped columns in extension views. + $node->safe_psql('postgres', 'UPDATE vstat_vm SET id = id + 10000'); + wait_for_updates('vstat_vm', 5000); + # Do not scan the heap here: that could prune the obsolete versions before + # VACUUM gets to count their removal. + is($node->safe_psql('postgres', + "SELECT pg_relation_size('vstat_vm') / current_setting('block_size')::bigint"), + $pages, 'updated versions fit on the original pages'); + is(counters('vstat_vm', 'frozen_page_marks_cleared, visible_page_marks_cleared'), + "0|$pages", 'extension view reports only all-visible clearings for unfrozen pages'); + + $verbose = vacuum_table('vstat_vm', '(FREEZE, VERBOSE)'); + my $visible = 2 * $pages; + is(counters('vstat_vm', 'tuples_deleted, pages_frozen, pages_all_visible'), + "5000|$pages|$visible", 'FREEZE reports removals, freezing and restored visibility separately'); + is(counters('vstat_vm', 'frozen_page_marks_cleared, visible_page_marks_cleared'), + "0|$pages", 'restoring VM flags adds no clearings'); + like($verbose, qr/pages with tuples frozen: \Q$pages\E\n/, + 'VERBOSE reports exactly one freeze per populated page'); + like($verbose, qr/pages marked all-visible: \Q$pages\E\n/, + 'VERBOSE reports this run\'s visibility work, not the cumulative total'); + is(counters('vstat_vm', 'freeze_age_vacuum_count'), '1', + 'FREEZE counts the vacuum made aggressive by the freeze age'); + like($verbose, qr/aggressive scan required by freeze age: yes/, + 'VERBOSE explains freeze-age-driven aggressive scanning'); + + $verbose = vacuum_table('vstat_vm', '(FREEZE, VERBOSE)'); + is(counters('vstat_vm', 'pages_frozen, pages_all_visible'), + "$pages|$visible", 'another FREEZE adds no freezing or visibility work'); + like($verbose, qr/pages with tuples frozen: 0\n/, + 'VERBOSE reports no repeated freezing'); + like($verbose, qr/pages marked all-visible: 0\n/, + 'VERBOSE reports no repeated visibility changes'); + + $node->safe_psql('postgres', 'DELETE FROM vstat_vm'); + wait_for_stats(q{ +SELECT n_tup_del = 5000 FROM pg_stat_all_tables_internal WHERE relname = 'vstat_vm' +}, 'DELETE report for frozen pages'); + is(counters('vstat_vm', 'frozen_page_marks_cleared, visible_page_marks_cleared'), + "$pages|$visible", 'extension view exposes both core VM counters with distinct totals'); +}; + +subtest 'append-optimized compaction' => sub { + for my $am ('ao_row', 'ao_column') + { + my $table = "vstat_$am"; + $node->safe_psql('postgres', qq{ +CREATE TABLE $table (id int, val int) USING $am; +CREATE INDEX ${table}_idx ON $table (id); +INSERT INTO $table SELECT i, i FROM generate_series(1, 10000) g(i); +DELETE FROM $table WHERE id % 2 = 0; +}); + # Preserve the zero-freeze-age case: AO still must not count a + # freeze-age vacuum, unlike its auxiliary heap relations. + my $verbose = vacuum_table($table, 'VERBOSE', 'SET vacuum_freeze_table_age = 0;'); + is(counters($table, 'tuples_deleted, dead_tuples, pages_frozen, pages_all_visible, freeze_age_vacuum_count'), + '5000|0|0|0|0', "$am compaction removes 5000 rows and has no heap VM or freezing work"); + # Compaction relocates surviving tuples, so all original index + # entries, including those of surviving rows, become obsolete. + is(index_counters("${table}_idx", 'tuples_deleted'), + '10000', "$am index cleanup removes the old TIDs"); + is(counters($table, 'total_vacuum_time > 0, total_autovacuum_time = 0, total_vacuum_delay_time = 0'), + 't|t|t', "$am compaction takes time without cost delay"); + my $bytes = counters($table, 'bytes_removed'); + my $segrel = $node->safe_psql('postgres', + "SELECT segrelid::regclass FROM pg_appendonly WHERE relid = '$table'::regclass"); + my $segs = $node->safe_psql('postgres', "SELECT count(*) FROM $segrel"); + is(counters($table, 'total_file_segs'), $segs, + "$am reports the remaining segment metadata entries"); + is($node->safe_psql('postgres', + "SELECT total_file_segs FROM gp_stat_vacuum_tables WHERE relname = '$table'"), + $segs, "$am cluster view exposes the segment snapshot in utility mode"); + like($verbose, qr/\Q$bytes bytes truncated; $segs file segments remain.\E/, + "$am VERBOSE byte and segment counts match the report"); + like($verbose, + qr/append-optimized table "\Q$table\E": vacuum statistics\nDETAIL: \Q0 dead tuples remain.\E\n\d+ bytes truncated; \d+ file segments remain\.\nelapsed: \d+\.\d{3} ms, cost-based delay: 0\.000 ms/, + "$am VERBOSE reports remaining dead tuples and accumulated phase time"); + is(counters($table, 'pages_deleted > 0, dead_pages, frozen_page_marks_cleared, visible_page_marks_cleared'), + 't|0|0|0', "$am reports freed space without heap page or VM counters"); + + # Without obsolete segment files, AO still runs index cleanup. + # Its time must be reported even if the AM returns no page statistics. + my $index_time = index_counters("${table}_idx", 'total_vacuum_time'); + vacuum_table($table); + is(counters($table, 'bytes_removed, total_file_segs'), "$bytes|$segs", + "$am idle vacuum neither adds reclaimed bytes nor sums segment counts"); + is(index_counters("${table}_idx", "tuples_deleted, total_vacuum_time > $index_time, total_vacuum_delay_time = 0"), + '10000|t|t', "$am reports index cleanup without deleting more TIDs"); + + $node->safe_psql('postgres', "DELETE FROM $table WHERE id <= 200"); + vacuum_table($table, '', 'SET gp_appendonly_compaction = off;'); + is(counters($table, 'tuples_deleted, dead_tuples'), + '5000|100', "$am counts hidden rows left when compaction is disabled"); + } +}; + +subtest 'AO truncate reports exact bytes beyond the 32-bit boundary' => sub { + my $block_size = $node->safe_psql('postgres', 'SHOW block_size'); + for my $am ('ao_row', 'ao_column') + { + my $table = "vstat_tail_$am"; + $node->safe_psql('postgres', qq{ +CREATE TABLE $table (id int) USING $am; +INSERT INTO $table SELECT generate_series(1, 10000); +CHECKPOINT; +}); + my $relpath = $node->safe_psql('postgres', + "SELECT pg_relation_filepath('$table')"); + my @files = grep { -f $_ && -s $_ } + glob($node->data_dir . '/' . $relpath . '*'); + my ($file) = grep { /\Q$relpath\E(?:\.\d+)?$/ } @files; + defined($file) or BAIL_OUT("no data file for $table"); + my $original_size = -s $file; + my $tail_size = 2**31 + 17; + # Model an aborted insert's tail in this disposable cluster with a + # sparse file. This exercises real truncate without writing 2 GiB. + open(my $fh, '+<', $file) or die "$file: $!"; + truncate($fh, $original_size + $tail_size) or die "truncate: $!"; + close($fh) or die "close: $!"; + vacuum_table($table); + is(-s $file, $original_size, "$am removes the physical tail"); + my $pages = int(($tail_size + $block_size - 1) / $block_size); + is(counters($table, 'bytes_removed, pages_deleted'), "$tail_size|$pages", + "$am preserves exact bytes and rounds the block equivalent up"); + my $segs = counters($table, 'total_file_segs'); + cmp_ok($segs, '>', 0, "$am has a segment snapshot"); + $node->restart; + is(counters($table, 'bytes_removed, pages_deleted, total_file_segs'), + "$tail_size|$pages|$segs", "$am counters and state survive restart"); + vacuum_table($table); + is(counters($table, 'bytes_removed, pages_deleted, total_file_segs'), + "$tail_size|$pages|$segs", "$am repeated vacuum does not recount the tail or segments"); + run_and_wait("SELECT vacuum_stats_reset('$table'::regclass::oid)"); + is(counters($table, 'bytes_removed, total_file_segs'), '0|0', + "$am relation reset clears both cumulative bytes and segment state"); + } +}; + +subtest 'SP-GiST does not recount reusable pages' => sub { + for my $am ('heap', 'ao_row', 'ao_column') + { + my $table = "vstat_spg_$am"; + $node->safe_psql('postgres', qq{ +CREATE TABLE $table (id int, p point) USING $am; +CREATE INDEX ${table}_idx ON $table USING spgist (p); +INSERT INTO $table SELECT i, point(i,i) FROM generate_series(1, 10000) g(i); +DELETE FROM $table; +}); + vacuum_table($table, '(INDEX_CLEANUP ON)'); + my $pages = index_counters("${table}_idx", 'pages_deleted'); + is(index_counters("${table}_idx", 'tuples_deleted'), '10000', + "$am SP-GiST removes the original index entries"); + vacuum_table($table, '(INDEX_CLEANUP ON)'); + is(index_counters("${table}_idx", 'pages_deleted'), $pages, + "$am SP-GiST does not count already empty pages again"); + is(index_counters("${table}_idx", 'bytes_removed'), '0', + "$am SP-GiST page reuse returns no bytes to the filesystem"); + } +}; + +subtest 'VACUUM FULL leaves the extended counters unchanged' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_full (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_full SELECT generate_series(1, 10000); +DELETE FROM vstat_full WHERE id % 2 = 0; +}); + vacuum_table('vstat_full'); + is(counters('vstat_full', 'tuples_deleted'), '5000', + 'plain vacuum establishes nonzero counters'); + $node->safe_psql('postgres', 'DELETE FROM vstat_full'); + wait_for_stats( + "SELECT n_tup_del = 10000 FROM pg_stat_all_tables_internal WHERE relname = 'vstat_full'", + 'DELETE report before VACUUM FULL'); + my $table_before = counters('vstat_full'); + my $index_before = index_counters('vstat_full_pkey'); + run_and_wait('VACUUM FULL vstat_full'); + is(counters('vstat_full'), $table_before, 'VACUUM FULL leaves table counters unchanged'); + is(index_counters('vstat_full_pkey'), $index_before, 'VACUUM FULL leaves index counters unchanged'); +}; + +subtest 'cost-based delay is part of total time' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_delay (id int) WITH (autovacuum_enabled = off); +INSERT INTO vstat_delay SELECT generate_series(1, 10000); +DELETE FROM vstat_delay; +}); + # Session-local settings do not leak into later scenarios. + my $verbose = vacuum_table('vstat_delay', 'VERBOSE', + 'SET track_cost_delay_timing = on; SET vacuum_cost_delay = 1; SET vacuum_cost_limit = 1;'); + is(counters('vstat_delay', 'total_vacuum_delay_time > 0 AND total_vacuum_delay_time <= total_vacuum_time'), + 't', 'the measured cost delay is positive and included in total time'); + my $delay = sprintf('%.3f', counters('vstat_delay', 'total_vacuum_delay_time')); + like($verbose, qr/elapsed: \d+\.\d{3} ms, cost-based delay: \Q$delay\E ms/, + 'VERBOSE reports elapsed time and the same measured cost delay'); +}; + +subtest 'index pages are not counted again by later vacuums' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_idx (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_idx SELECT generate_series(1, 100000); +DELETE FROM vstat_idx WHERE id <= 90000; +}); + vacuum_table('vstat_idx'); + my $pages = index_counters('vstat_idx_pkey', 'pages_deleted'); + cmp_ok($pages, '>', 0, 'the first vacuum deleted index pages'); + $node->safe_psql('postgres', 'DELETE FROM vstat_idx WHERE id > 99000'); + vacuum_table('vstat_idx'); + is(index_counters('vstat_idx_pkey', 'tuples_deleted'), '91000', + 'the next vacuum removed exactly 1000 more index entries'); + is(index_counters('vstat_idx_pkey', "pages_deleted >= $pages AND pages_deleted - $pages < $pages / 2"), + 't', 'deleted-page totals accumulate without recounting the first run'); +}; + +subtest 'reset preserves ordinary statistics and other relations' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_reset (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +CREATE TABLE vstat_keep (id int) WITH (autovacuum_enabled = off); +INSERT INTO vstat_reset SELECT generate_series(1, 1000); +INSERT INTO vstat_keep SELECT generate_series(1, 1000); +DELETE FROM vstat_keep; +}); + vacuum_table('vstat_keep'); + vacuum_table('vstat_reset', 'FREEZE'); + $node->safe_psql('postgres', 'UPDATE vstat_reset SET id = id + 1000'); + wait_for_updates('vstat_reset', 1000); + vacuum_table('vstat_reset', 'FREEZE'); + is(counters('vstat_reset', + 'tuples_deleted, frozen_page_marks_cleared > 0, visible_page_marks_cleared > 0'), + '1000|t|t', 'populate removal and VM revision counters before reset'); + my $keep_before = counters('vstat_keep'); + my $index_before = index_counters('vstat_reset_pkey'); + my $db_before = database_counters(); + my $ordinary_query = "SELECT vacuum_count, n_tup_upd FROM pg_stat_all_tables_internal WHERE relname = 'vstat_reset'"; + my $ordinary_before = $node->safe_psql('postgres', $ordinary_query); + my $reset_before = counters('vstat_reset'); + + my ($result, $stdout, $stderr) = $node->psql('postgres', + 'SELECT vacuum_stats_reset(0::oid)'); + is($result, 3, 'an invalid relation OID is rejected'); + like($stderr, qr/invalid relation OID: 0/, 'invalid OID has a specific error'); + run_and_wait('SELECT vacuum_stats_reset(NULL::oid)'); + is(counters('vstat_reset'), $reset_before, + 'invalid and NULL OIDs do not reset a relation'); + is(database_counters(), $db_before, + 'invalid and NULL OIDs do not reset database totals'); + + run_and_wait("SELECT vacuum_stats_reset('vstat_reset'::regclass::oid)"); + is(counters('vstat_reset', $all_zero), 't', 'relation reset clears every vacuum counter including VM clearing counters'); + is(counters('vstat_keep'), $keep_before, 'relation reset preserves another table'); + is(index_counters('vstat_reset_pkey'), $index_before, 'relation reset preserves the index counters'); + is(database_counters(), $db_before, 'relation reset preserves database totals'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'relation reset preserves ordinary statistics'); + + # Refill the reset relation, including both VM revision counters, + # before checking the database-wide reset. + $node->safe_psql('postgres', 'UPDATE vstat_reset SET id = id + 1000'); + wait_for_updates('vstat_reset', 2000); + vacuum_table('vstat_reset', 'FREEZE'); + is(counters('vstat_reset', + 'tuples_deleted, frozen_page_marks_cleared > 0, visible_page_marks_cleared > 0'), + '1000|t|t', 'the counters are nonzero again before database reset'); + $ordinary_before = $node->safe_psql('postgres', $ordinary_query); + run_and_wait('SELECT vacuum_stats_reset()'); + is($node->safe_psql('postgres', "SELECT bool_and($all_zero) FROM pg_stat_vacuum_tables"), + 't', 'database reset clears all table counters'); + is($node->safe_psql('postgres', 'SELECT bool_and(total_file_segs = 0) FROM pg_stat_vacuum_tables'), + 't', 'database reset clears AO segment snapshots'); + is($node->safe_psql('postgres', "SELECT bool_and($all_zero) FROM pg_stat_vacuum_indexes"), + 't', 'database reset clears all index counters'); + is(database_counters($all_zero), 't', 'database reset clears database totals'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'database reset preserves ordinary statistics'); + + +}; + +subtest 'database totals do not count index work twice' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_db (id int PRIMARY KEY, val int) WITH (autovacuum_enabled = off); +CREATE INDEX ON vstat_db (val); +INSERT INTO vstat_db SELECT i, i FROM generate_series(1, 10000) g(i); +DELETE FROM vstat_db WHERE id % 2 = 0; +}); + run_and_wait('SELECT vacuum_stats_reset()'); + vacuum_table('vstat_db'); + is($node->safe_psql('postgres', + "SELECT sum(tuples_deleted) FROM pg_stat_vacuum_indexes WHERE relname = 'vstat_db'"), + '10000', 'both indexes report 5000 removed entries'); + is(database_counters('tuples_deleted'), '5000', 'database totals count only heap tuples'); + is(database_counters(), counters('vstat_db'), + 'all database vacuum counters match the only table vacuumed since reset'); +}; + +subtest 'statistics snapshots release their vacuum counters' => sub { + # Keep the last snapshot alive until the context count is read. + # Outside this transaction its context would already have been freed. + is($node->safe_psql('postgres', q{ +BEGIN; +DO $$ +BEGIN + FOR i IN 1..50 LOOP + PERFORM pg_stat_clear_snapshot(); + PERFORM count(*) FROM pg_stat_vacuum_tables WHERE tuples_deleted > 0; + END LOOP; +END +$$; +SELECT count(*) FROM pg_backend_memory_contexts +WHERE name = 'Databases hash'; +COMMIT; +}), '1', 'only the current snapshot owns statistics entries with vacuum counters'); +}; + +subtest 'reset targets shared catalogs independently' => sub { + # Shared catalogs have a separate collector entry, not the current DB's. + # Create actual dead index entries instead of timing an empty cleanup. + $node->safe_psql('postgres', 'CREATE DATABASE vstat_shared_reset'); + $node->safe_psql('postgres', 'DROP DATABASE vstat_shared_reset'); + vacuum_table('pg_database', '(FREEZE, INDEX_CLEANUP ON)'); + is(counters('pg_database', 'total_vacuum_time > 0'), 't', + 'populate vacuum counters for a shared catalog'); + my $shared_before = counters('pg_database'); + run_and_wait('SELECT vacuum_stats_reset()'); + is(counters('pg_database'), $shared_before, + 'database reset leaves shared catalog counters alone'); + + vacuum_table('vstat_reset'); + my $reset_before = counters('vstat_reset'); + my $db_before = database_counters(); + run_and_wait("SELECT vacuum_stats_reset('pg_database'::regclass::oid)"); + is(counters('pg_database', $all_zero), 't', + 'relation reset reaches the shared catalog entry'); + is(counters('vstat_reset'), $reset_before, + 'shared relation reset leaves current-database relations alone'); + is(database_counters(), $db_before, + 'shared relation reset leaves current-database totals alone'); + + # Indexes are separate reset targets, including shared catalog indexes. + is(index_counters('pg_database_oid_index', 'tuples_deleted > 0'), 't', + 'resetting the shared table preserves its index counters'); + run_and_wait(q{ +SELECT vacuum_stats_reset(indexrelid) FROM pg_index +WHERE indrelid = 'pg_database'::regclass +}); + is(index_counters('pg_database_oid_index', $all_zero), 't', + 'relation reset also reaches shared index counters'); +}; + +subtest 'disabling tracking omits vacuum counters and preserves ordinary statistics' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_startup (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_startup SELECT generate_series(1, 1000); +}); + vacuum_table('vstat_startup', 'FREEZE'); + $node->safe_psql('postgres', 'UPDATE vstat_startup SET id = id + 1000'); + wait_for_updates('vstat_startup', 1000); + vacuum_table('vstat_startup', 'FREEZE'); + is(counters('vstat_startup', + 'tuples_deleted, frozen_page_marks_cleared > 0, visible_page_marks_cleared > 0'), + '1000|t|t', 'populate vacuum and VM revision counters before disabling'); + my $ordinary_query = "SELECT vacuum_count, n_tup_upd FROM pg_stat_all_tables_internal WHERE relname = 'vstat_startup'"; + my $ordinary_before = $node->safe_psql('postgres', $ordinary_query); + my $vm_before = counters('vstat_startup', $vm_columns); + my $timing_columns = 'total_vacuum_time, total_autovacuum_time, total_vacuum_delay_time, total_autovacuum_delay_time'; + my $timing_before = counters('vstat_startup', $timing_columns); + my $db_vm_before = database_counters($vm_columns); + my $on_bytes = snapshot_bytes(); + $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = off'); + $node->restart; + is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'off', + 'restart disables tracking'); + is(counters('vstat_startup', $vm_columns), $vm_before, + 'disabling vacuum tracking preserves relation VM counters'); + is(counters('vstat_startup', $timing_columns), $timing_before, + 'disabling extended tracking preserves core vacuum timing'); + is(database_counters($vm_columns), $db_vm_before, + 'disabling vacuum tracking preserves database VM totals'); + is(counters('vstat_startup', $vacuum_zero), 't', 'disabled table counters read as zero'); + is(index_counters('vstat_startup_pkey', $vacuum_zero), 't', 'disabled index counters read as zero'); + is(database_counters($vacuum_zero), 't', 'disabled database counters read as zero'); + is($node->safe_psql('postgres', 'SELECT bool_and(total_file_segs = 0) FROM pg_stat_vacuum_tables'), + 't', 'disabled AO segment snapshots read as zero'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'reading statistics with tracking disabled preserves ordinary counters'); + cmp_ok(snapshot_bytes(), '<', $on_bytes, + 'disabled snapshots omit vacuum storage but retain ordinary VM counters'); + run_and_wait("SELECT vacuum_stats_reset('vstat_startup'::regclass::oid)"); + is(counters('vstat_startup', $vm_columns), '0|0', + 'relation reset clears VM counters with vacuum tracking off'); + is(database_counters($vm_columns), $db_vm_before, + 'relation reset preserves database VM totals with vacuum tracking off'); + run_and_wait('SELECT vacuum_stats_reset()'); + is(database_counters($vm_columns), '0|0', + 'database reset clears VM totals with vacuum tracking off'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'vacuum reset while disabled leaves ordinary counters intact'); + $node->safe_psql('postgres', 'ALTER SYSTEM SET track_vacuum_statistics = on'); + $node->restart; + is(counters('vstat_startup', $vacuum_zero), 't', + 're-enabling does not restore discarded table vacuum counters'); + is(database_counters($vacuum_zero), 't', + 're-enabling does not restore discarded database vacuum counters'); + is($node->safe_psql('postgres', $ordinary_query), $ordinary_before, + 'ordinary statistics survive both changes in record size'); + vacuum_table('vstat_startup', 'FREEZE'); + is(counters('vstat_startup', 'total_vacuum_time > 0'), 't', + 'vacuum reporting resumes after tracking is enabled again'); +}; + +subtest 'clean restart preserves statistics; crash recovery resets them' => sub { + $node->safe_psql('postgres', q{ +CREATE TABLE vstat_restart (id int PRIMARY KEY) WITH (autovacuum_enabled = off); +INSERT INTO vstat_restart SELECT generate_series(1, 1000); +DELETE FROM vstat_restart WHERE id % 2 = 0; +}); + vacuum_table('vstat_restart', 'FREEZE'); + is(counters('vstat_restart', 'tuples_deleted'), '500', 'table counters are populated before restart'); + is(index_counters('vstat_restart_pkey', 'tuples_deleted'), '500', 'index counters are populated before restart'); + my $table_before = counters('vstat_restart'); + my $index_before = index_counters('vstat_restart_pkey'); + my $db_before = database_counters(); + $node->restart; + is(counters('vstat_restart'), $table_before, 'table counters survive a clean restart'); + is(index_counters('vstat_restart_pkey'), $index_before, 'index counters survive a clean restart'); + is(database_counters(), $db_before, 'database counters survive a clean restart'); + $node->stop('immediate'); + $node->start; + is(counters('vstat_restart', $all_zero), 't', 'crash recovery resets table counters'); + is(index_counters('vstat_restart_pkey', $all_zero), 't', 'crash recovery resets index counters'); + is(database_counters($all_zero), 't', 'crash recovery resets database counters'); +}; + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/t/002_index_vacuum_time.pl b/contrib/vacuum_stats/t/002_index_vacuum_time.pl new file mode 100644 index 00000000000..ff41161e7fa --- /dev/null +++ b/contrib/vacuum_stats/t/002_index_vacuum_time.pl @@ -0,0 +1,252 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Test the total_vacuum_time counter of pg_stat_vacuum_indexes: the time vacuum +# spent processing each index, accumulated over bulkdelete and cleanup passes, +# mirroring the table-level counter in pg_stat_vacuum_tables. + +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('main'); +$node->init; +$node->append_conf( + 'postgresql.conf', qq[ +autovacuum = off +track_cost_delay_timing = on +]); +$node->start; +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats'); + +# The index has to span enough pages for its vacuum passes to exceed +# vacuum_cost_limit on buffer hits alone (its pages are still dirty from the +# load, so dirtying them costs nothing). Scale the row count with the block +# size, so that builds with larger blocks (32kB in Cloudberry) keep the page +# count of the default 8kB. +my $nrows = 100000 * + ($node->safe_psql('postgres', 'SHOW block_size') / 8192); + +# Enough rows that the bulkdelete pass over the index takes measurable time; +# a small vacuum_cost_delay adds deterministic delay time on fast machines. +$node->safe_psql( + 'postgres', qq[ +CREATE TABLE vactime_t (id int PRIMARY KEY, v text) WITH (autovacuum_enabled = off); +INSERT INTO vactime_t SELECT g, repeat('x', 10) FROM generate_series(1, $nrows) g; +DELETE FROM vactime_t WHERE id % 2 = 0; +]); +$node->safe_psql( + 'postgres', qq[ +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 200; +VACUUM vactime_t; +]); + +# The collector receives the ordinary vacuum and index-pass reports asynchronously. +$node->poll_query_until('postgres', q{ +SELECT total_vacuum_time > 0 AND total_vacuum_delay_time > 0 +FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey' +}) or BAIL_OUT('index timing report did not reach the collector'); +$node->poll_query_until('postgres', q{ +SELECT total_vacuum_time > 0 AND total_vacuum_delay_time > 0 +FROM pg_stat_vacuum_database WHERE datname = current_database() +}) or BAIL_OUT('database timing report did not reach the collector'); + +# The upper bound guards against garbage such as an epoch-based elapsed time +# leaking into the counter. +is( $node->safe_psql( + 'postgres', qq[ +SELECT total_vacuum_time > 0 AND total_vacuum_time < 600000 + FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey']), + 't', + 'total_vacuum_time advanced sanely for the index in pg_stat_vacuum_indexes'); + +is( $node->safe_psql( + 'postgres', qq[ +SELECT total_vacuum_delay_time > 0 AND total_vacuum_delay_time <= total_vacuum_time + FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey']), + 't', + 'total_vacuum_delay_time advanced and not above total_vacuum_time'); + +is( $node->safe_psql( + 'postgres', qq[ +SELECT total_autovacuum_time = 0 FROM pg_stat_vacuum_indexes + WHERE indexrelname = 'vactime_t_pkey']), + 't', + 'manual vacuum did not count into total_autovacuum_time'); + +# The same run must have accumulated into the database-wide totals. +is( $node->safe_psql( + 'postgres', qq[ +SELECT total_vacuum_time > 0 AND total_vacuum_time < 600000 + AND total_vacuum_delay_time > 0 + AND total_vacuum_delay_time <= total_vacuum_time + FROM pg_stat_vacuum_database WHERE datname = current_database()]), + 't', + 'database-wide vacuum times advanced sanely in pg_stat_vacuum_database'); + +# The database total is the sum of table runs, which already include index +# work. Adding the positive per-index time again would break this equality. +# Cast each millisecond value to numeric before summing to avoid floating-point +# rounding differences. Shared relations belong to the database with OID zero. +is($node->safe_psql('postgres', q{ +SELECT d.total_vacuum_time::numeric = t.elapsed + AND d.total_vacuum_delay_time::numeric = t.delay +FROM pg_stat_vacuum_database d CROSS JOIN ( + SELECT sum(s.total_vacuum_time::numeric) AS elapsed, + sum(s.total_vacuum_delay_time::numeric) AS delay + FROM pg_stat_vacuum_tables s JOIN pg_class c ON c.oid = s.relid + WHERE NOT c.relisshared +) t +WHERE d.datname = current_database() +}), 't', 'database totals include index work exactly once'); + +# Turning off delay timing must preserve the accumulated delay while elapsed +# time keeps advancing. Keep cost delays enabled to exercise the timing GUC. +my $manual_times = q{ +SELECT total_vacuum_time, total_vacuum_delay_time +FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey' +}; +my ($elapsed_before, $delay_before) = + split /\|/, $node->safe_psql('postgres', $manual_times); +$node->safe_psql('postgres', q{ +SET track_cost_delay_timing = off; +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 200; +DELETE FROM vactime_t WHERE id % 4 = 1; +VACUUM (INDEX_CLEANUP ON) vactime_t; +}); +$node->poll_query_until('postgres', + "SELECT total_vacuum_time > $elapsed_before FROM pg_stat_vacuum_indexes WHERE indexrelname = 'vactime_t_pkey'") + or BAIL_OUT('index elapsed time did not advance with delay timing disabled'); +my ($elapsed_after, $delay_after) = + split /\|/, $node->safe_psql('postgres', $manual_times); +is($delay_after, $delay_before, 'disabled delay timing preserves the index delay total'); + +# Exercise ANALYZE and VACUUM in the same backend, where the process-local +# delay accumulator survives between statements. Check their SQL totals +# independently rather than inferring one from the other. +# Drain earlier asynchronous vacuum reports before checking what ANALYZE +# alone changes. +$node->restart; +my $table_manual_times = q{ +SELECT total_vacuum_time, total_vacuum_delay_time +FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t' +}; +my $db_manual_times = q{ +SELECT total_vacuum_time, total_vacuum_delay_time +FROM pg_stat_vacuum_database WHERE datname = current_database() +}; +my @timing_queries = ($table_manual_times, $manual_times, $db_manual_times); +my @timing_labels = ('table', 'index', 'database'); +my @before_analyze = map { $node->safe_psql('postgres', $_) } @timing_queries; +$node->safe_psql('postgres', q{ +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 200; +ANALYZE vactime_t; +}); +$node->poll_query_until('postgres', q{ +SELECT total_analyze_time > 0 FROM pg_stat_vacuum_tables +WHERE relname = 'vactime_t' +}) or BAIL_OUT('ANALYZE timing report did not reach the collector'); +for my $i (0..2) +{ + is($node->safe_psql('postgres', $timing_queries[$i]), $before_analyze[$i], + "ANALYZE preserves $timing_labels[$i] vacuum elapsed and delay times"); +} +my $analyze_before = $node->safe_psql('postgres', q{ +SELECT total_analyze_time FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t' +}); +my ($table_elapsed, $table_delay) = split /\|/, $before_analyze[0]; +$node->safe_psql('postgres', q{ +SET vacuum_cost_delay = '1ms'; +SET vacuum_cost_limit = 200; +ANALYZE vactime_t; +VACUUM (ANALYZE, DISABLE_PAGE_SKIPPING, INDEX_CLEANUP ON) vactime_t; +ANALYZE vactime_t; +}); +$node->poll_query_until('postgres', qq{ +SELECT total_vacuum_time > $table_elapsed + AND total_vacuum_delay_time > $table_delay + AND total_analyze_time > $analyze_before +FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t' +}) or BAIL_OUT('VACUUM ANALYZE did not report both maintenance phases'); +is($node->safe_psql('postgres', q{ +SELECT total_vacuum_delay_time <= total_vacuum_time + AND total_autovacuum_time = 0 AND total_autoanalyze_time = 0 +FROM pg_stat_vacuum_tables WHERE relname = 'vactime_t' +}), 't', 'VACUUM ANALYZE records separate manual timing totals'); + +# Cloudberry autovacuums catalogs. Churn comments to create dead catalog +# tuples and index entries, then wait for a real worker to process them. +$node->safe_psql('postgres', q{ +CREATE TABLE vactime_autovacuum (id int) WITH (autovacuum_enabled = off); +COMMENT ON TABLE vactime_autovacuum IS 'initial'; +VACUUM (INDEX_CLEANUP ON) pg_description; +}); +$node->poll_query_until('postgres', q{ +SELECT total_vacuum_time > 0 FROM pg_stat_vacuum_tables WHERE relname = 'pg_description' +}) or BAIL_OUT('catalog vacuum timing did not reach the collector'); +my $table_time = "SELECT total_vacuum_time, total_autovacuum_time FROM pg_stat_vacuum_tables WHERE relname = 'pg_description'"; +my $index_time = "SELECT total_vacuum_time, total_autovacuum_time FROM pg_stat_vacuum_indexes WHERE indexrelname = 'pg_description_o_c_o_index'"; +my $db_time = "SELECT total_vacuum_time, total_autovacuum_time FROM pg_stat_vacuum_database WHERE datname = current_database()"; +my @manual_before = map { (split /\|/, $node->safe_psql('postgres', $_))[0] } + ($table_time, $index_time, $db_time); +$node->safe_psql('postgres', q{ +DO $$ BEGIN + FOR i IN 1..200 LOOP + EXECUTE 'COMMENT ON TABLE vactime_autovacuum IS NULL'; + EXECUTE format('COMMENT ON TABLE vactime_autovacuum IS %L', i::text); + END LOOP; +END $$; +ALTER SYSTEM SET autovacuum_naptime = '1s'; +ALTER SYSTEM SET autovacuum_vacuum_threshold = 0; +ALTER SYSTEM SET autovacuum_vacuum_scale_factor = 0; +ALTER SYSTEM SET autovacuum_vacuum_insert_threshold = -1; +ALTER SYSTEM SET autovacuum = on; +}); +$node->reload; +$node->poll_query_until('postgres', q{ +SELECT total_autovacuum_time > 0 FROM pg_stat_vacuum_tables WHERE relname = 'pg_description' +}) or BAIL_OUT('autovacuum did not report catalog time'); +$node->poll_query_until('postgres', q{ +SELECT total_autovacuum_time > 0 FROM pg_stat_vacuum_indexes +WHERE indexrelname = 'pg_description_o_c_o_index' +}) or BAIL_OUT('autovacuum did not report catalog index time'); +$node->safe_psql('postgres', 'ALTER SYSTEM SET autovacuum = off'); +# Drain the collector on a clean shutdown before comparing exact totals. +# A worker disappearing from pg_stat_activity alone does not prove that +# its final UDP reports have reached the collector. +$node->restart; +my @auto_before; +my @queries = ($table_time, $index_time, $db_time); +my @labels = ('table', 'index', 'database'); +for my $i (0..2) +{ + my ($manual, $automatic) = split /\|/, $node->safe_psql('postgres', $queries[$i]); + is($manual, $manual_before[$i], "autovacuum preserves manual $labels[$i] time"); + cmp_ok($automatic, '>', 0, "autovacuum records separate $labels[$i] time"); + push @auto_before, $automatic; +} +$node->safe_psql('postgres', 'VACUUM (INDEX_CLEANUP ON) pg_description'); +$node->poll_query_until('postgres', + "SELECT total_vacuum_time > $manual_before[0] FROM pg_stat_vacuum_tables WHERE relname = 'pg_description'") + or BAIL_OUT('manual catalog vacuum did not report new time'); +for my $i (0..2) +{ + my ($manual, $automatic) = split /\|/, $node->safe_psql('postgres', $queries[$i]); + cmp_ok($manual, '>', $manual_before[$i], "manual vacuum adds $labels[$i] time"); + is($automatic, $auto_before[$i], "manual vacuum preserves autovacuum $labels[$i] time"); +} +my @before_restart = map { $node->safe_psql('postgres', $_) } @queries; +$node->restart; +for my $i (0..2) +{ + is($node->safe_psql('postgres', $queries[$i]), $before_restart[$i], + "both $labels[$i] timing counters survive restart"); +} + +$node->stop; + +done_testing(); diff --git a/contrib/vacuum_stats/t/003_vacuum_failsafe.pl b/contrib/vacuum_stats/t/003_vacuum_failsafe.pl new file mode 100644 index 00000000000..a41cd054b53 --- /dev/null +++ b/contrib/vacuum_stats/t/003_vacuum_failsafe.pl @@ -0,0 +1,113 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Test that vacuums entering the wraparound failsafe mode are counted in +# pg_stat_vacuum_tables.vacuum_failsafe_count and aggregated per database in +# pg_stat_vacuum_database.vacuum_failsafe_count. +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('main'); +$node->init; +# The failsafe cutoff is clamped to 1.05 * autovacuum_freeze_max_age, so use +# the minimum allowed value to keep the number of XIDs to burn small. +$node->append_conf( + 'postgresql.conf', qq[ +autovacuum = off +autovacuum_freeze_max_age = 100000 +]); +$node->start; +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats'); + +$node->safe_psql( + 'postgres', qq[ + CREATE TABLE tab_failsafe (i int); + INSERT INTO tab_failsafe SELECT generate_series(1, 100); +]); + +# A vacuum without failsafe pressure must not bump the counter. +$node->safe_psql('postgres', 'VACUUM FREEZE tab_failsafe;'); +$node->poll_query_until('postgres', q{ +SELECT vacuum_count = 1 FROM pg_stat_all_tables_internal WHERE relname = 'tab_failsafe' +}) or BAIL_OUT('normal vacuum report did not reach the collector'); +my $count = $node->safe_psql('postgres', + q[SELECT vacuum_failsafe_count FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe';] +); +is($count, '0', 'aggressive VACUUM FREEZE does not count as failsafe'); + +# Age the table past 1.05 * autovacuum_freeze_max_age: burn XIDs with +# aborted subtransactions (each aborted subxact consumes an assigned XID). +$node->safe_psql( + 'postgres', qq[ + CREATE TABLE burn_xids (i int); + DO \$\$ + BEGIN + FOR i IN 1..110000 LOOP + BEGIN + INSERT INTO burn_xids VALUES (1); + RAISE EXCEPTION 'burn'; + EXCEPTION WHEN OTHERS THEN + END; + END LOOP; + END \$\$; +]); + +# vacuum_failsafe_age = 0 makes the (clamped) cutoff kick in immediately. +$node->safe_psql( + 'postgres', qq[ + SET vacuum_failsafe_age = 0; + SET vacuum_multixact_failsafe_age = 0; + VACUUM tab_failsafe; +]); + +$node->poll_query_until('postgres', q{ +SELECT vacuum_failsafe_count = 1 FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe' +}) or BAIL_OUT('failsafe vacuum report did not reach the collector'); + +$count = $node->safe_psql('postgres', + q[SELECT vacuum_failsafe_count FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe';] +); +is($count, '1', 'failsafe vacuum counted in pg_stat_vacuum_tables'); + +my $db_count = $node->safe_psql('postgres', + q[SELECT vacuum_failsafe_count FROM pg_stat_vacuum_database WHERE datname = 'postgres';] +); +is($db_count, '1', 'failsafe vacuum counted in pg_stat_vacuum_database'); + +# Once the table has been frozen, another vacuum must not count the same +# failsafe event again. Wait for that run's report, not for an unchanged value. +$node->safe_psql('postgres', 'VACUUM tab_failsafe'); +$node->poll_query_until('postgres', q{ +SELECT vacuum_count = 3 FROM pg_stat_all_tables_internal WHERE relname = 'tab_failsafe' +}) or BAIL_OUT('subsequent vacuum report did not reach the collector'); +is($node->safe_psql('postgres', q{ +SELECT t.vacuum_failsafe_count, d.vacuum_failsafe_count +FROM pg_stat_vacuum_tables t CROSS JOIN pg_stat_vacuum_database d +WHERE t.relname = 'tab_failsafe' AND d.datname = current_database() +}), '1|1', 'subsequent ordinary vacuum does not count the failsafe again'); + +$node->restart; +is($node->safe_psql('postgres', q{ +SELECT t.vacuum_failsafe_count, d.vacuum_failsafe_count +FROM pg_stat_vacuum_tables t CROSS JOIN pg_stat_vacuum_database d +WHERE t.relname = 'tab_failsafe' AND d.datname = current_database() +}), '1|1', 'relation and database failsafe counts survive a clean restart'); + +$node->safe_psql('postgres', + q{SELECT pg_stat_reset_single_table_counters('tab_failsafe'::regclass)}); +$node->poll_query_until('postgres', q{ +SELECT vacuum_failsafe_count = 0 FROM pg_stat_vacuum_tables WHERE relname = 'tab_failsafe' +}) or BAIL_OUT('relation failsafe counter was not reset'); +is($node->safe_psql('postgres', q{ +SELECT vacuum_failsafe_count FROM pg_stat_vacuum_database WHERE datname = current_database() +}), '1', 'resetting a relation preserves the database failsafe total'); + +$node->safe_psql('postgres', 'SELECT pg_stat_reset()'); +ok($node->poll_query_until('postgres', q{ +SELECT vacuum_failsafe_count = 0 FROM pg_stat_vacuum_database WHERE datname = current_database() +}), 'database reset clears the failsafe total'); + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/t/004_visibility_map_stats.pl b/contrib/vacuum_stats/t/004_visibility_map_stats.pl new file mode 100644 index 00000000000..66256234b09 --- /dev/null +++ b/contrib/vacuum_stats/t/004_visibility_map_stats.pl @@ -0,0 +1,216 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# VM flag clearings are ordinary relation/database statistics. This test +# reads them through the vacuum_stats extension without changing core views. +# Adapt the v44 VM stability scenario to the collector used by Cloudberry: +# https://www.postgresql.org/message-id/attachment/204710/v44-0009-Track-table-VM-stability.patch +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('vm_stats'); +$node->init; +$node->append_conf('postgresql.conf', q{ +autovacuum = off +track_counts = on +}); +$node->start; +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats'); + +sub wait_for_stats +{ + my ($sql, $description) = @_; + $node->poll_query_until('postgres', $sql) + or BAIL_OUT("timed out waiting for $description"); +} + +my $columns = 'frozen_page_marks_cleared, visible_page_marks_cleared'; +my $table_query = "SELECT $columns FROM pg_stat_vacuum_tables WHERE relname = 'vm_heap'"; +my $visible_query = "SELECT $columns FROM pg_stat_vacuum_tables WHERE relname = 'vm_visible'"; +my $db_query = "SELECT $columns FROM pg_stat_vacuum_database WHERE datname = current_database()"; + +sub populated_pages +{ + my ($table) = @_; + return $node->safe_psql('postgres', + "SELECT count(DISTINCT split_part(ctid::text, ',', 1)) FROM $table"); +} + +$node->safe_psql('postgres', q{ +CREATE TABLE vm_heap (id int PRIMARY KEY) WITH (fillfactor = 70); +INSERT INTO vm_heap SELECT generate_series(1, 1000); +CREATE TABLE vm_visible (id int); +INSERT INTO vm_visible SELECT generate_series(1, 1000); +}); +wait_for_stats(q{ +SELECT n_tup_ins = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'insert report'); +$node->safe_psql('postgres', 'VACUUM FREEZE vm_heap'); +wait_for_stats(q{ +SELECT vacuum_count = 1 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'vacuum report'); +# A normal vacuum marks the new tuples visible, without freezing their XIDs. +$node->safe_psql('postgres', q{ +SET vacuum_freeze_min_age = 1000000000; +SET vacuum_freeze_table_age = 1000000000; +VACUUM vm_visible; +}); +wait_for_stats(q{ +SELECT vacuum_count = 1 AND n_tup_ins = 1000 +FROM pg_stat_all_tables_internal WHERE relname = 'vm_visible' +}, 'non-freezing vacuum and insert reports'); + +# Drain reports from table/index creation before taking the database baseline. +$node->restart; +is($node->safe_psql('postgres', $table_query), '0|0', + 'setting VM flags does not count as clearing them'); +my ($db_frozen, $db_visible) = split /\|/, + $node->safe_psql('postgres', $db_query); +my $pages = populated_pages('vm_heap'); +cmp_ok($pages, '>', 0, 'test table has populated heap pages'); + +# Count each physical transition once, even though UPDATE touches many tuples +# on each page. DML and VM counters travel in the same collector message, so +# wait for the DML report before checking exact values, including zeroes. +$node->safe_psql('postgres', 'UPDATE vm_heap SET id = id + 1000'); +wait_for_stats(q{ +SELECT n_tup_upd = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'first update report'); +is($node->safe_psql('postgres', $table_query), "$pages|$pages", + 'UPDATE clears one all-visible and all-frozen mark per populated page'); +is($node->safe_psql('postgres', qq{ +SELECT frozen_page_marks_cleared - $db_frozen, + visible_page_marks_cleared - $db_visible +FROM pg_stat_vacuum_database WHERE datname = current_database() +}), "$pages|$pages", 'database totals include exactly the table transitions'); + +$node->safe_psql('postgres', 'UPDATE vm_heap SET id = id + 1000'); +wait_for_stats(q{ +SELECT n_tup_upd = 2000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'second update report'); +is($node->safe_psql('postgres', $table_query), "$pages|$pages", + 'updates on pages whose marks are already clear add nothing'); +is($node->safe_psql('postgres', qq{ +SELECT pg_stat_get_frozen_page_marks_cleared('vm_heap'::regclass), + pg_stat_get_visible_page_marks_cleared('vm_heap'::regclass) +}), "$pages|$pages", 'extension getters and table view expose the same counters'); + +# Distinguish the two counters and verify that rolling back DML does not undo +# the physical clearing of VM bits. Utility-mode SELECT FOR UPDATE takes a +# table lock, so use unfrozen tuples to exercise independent bit accounting. +my $visible_pages = populated_pages('vm_visible'); +cmp_ok($visible_pages, '>', 0, 'unfrozen table has populated heap pages'); +$node->safe_psql('postgres', 'BEGIN; DELETE FROM vm_visible; ROLLBACK;'); +wait_for_stats(q{ +SELECT n_tup_del = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_visible' +}, 'aborted delete report'); +is($node->safe_psql('postgres', 'SELECT count(*) FROM vm_visible'), '1000', + 'aborted DELETE preserves the rows'); +is($node->safe_psql('postgres', $visible_query), "0|$visible_pages", + 'aborted DELETE counts all-visible clearings without inventing all-frozen clearings'); +is($node->safe_psql('postgres', qq{ +SELECT frozen_page_marks_cleared - $db_frozen, + visible_page_marks_cleared - $db_visible +FROM pg_stat_vacuum_database WHERE datname = current_database() +}), $pages . '|' . ($pages + $visible_pages), + 'database totals distinguish the bits and include aborted DML'); + +# Re-establish flags, then modify the table while a reader holds both an MVCC +# snapshot and a statistics snapshot. The statistics snapshot postpones +# observing the clearings; the VM bits themselves are cleared by DELETE, +# before the reader commits. +$node->safe_psql('postgres', 'VACUUM FREEZE vm_heap'); +wait_for_stats(q{ +SELECT vacuum_count = 2 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'second vacuum report'); +is($node->safe_psql('postgres', $table_query), "$pages|$pages", + 'restoring VM flags does not increase clearing counters'); +my $refrozen_pages = populated_pages('vm_heap'); +my $cleared = $pages + $refrozen_pages; + +my ($in, $out) = ('', ''); +my $timer = IPC::Run::timeout($TestLib::timeout_default); +my $reader = $node->background_psql('postgres', \$in, \$out, $timer); +my $reader_query = sub { + my ($sql) = @_; + $out = ''; + $in = "$sql;\n\\echo vm_query_done\n"; + pump_until($reader, $timer, \$out, qr/^vm_query_done\r?$/m) + or BAIL_OUT('reader did not complete its query'); + $out =~ s/\r//g; + $out =~ s/^vm_query_done\n?//m; + $out =~ s/^\n+|\n+$//g; + return $out; +}; +is($reader_query->("BEGIN ISOLATION LEVEL REPEATABLE READ; $table_query"), + "$pages|$pages", 'reader caches the pre-delete statistics'); +is($reader_query->('SELECT count(*) FROM vm_heap'), '1000', + 'reader holds an MVCC snapshot of the rows'); + +$node->safe_psql('postgres', 'DELETE FROM vm_heap'); +wait_for_stats(q{ +SELECT n_tup_del = 1000 FROM pg_stat_all_tables_internal WHERE relname = 'vm_heap' +}, 'concurrent delete report'); +is($node->safe_psql('postgres', $table_query), "$cleared|$cleared", + 'DELETE counts fresh VM clearings while the reader transaction is open'); +is($reader_query->($table_query), "$pages|$pages", + 'cached statistics retain their earlier values'); +is($reader_query->("SELECT pg_stat_clear_snapshot(); $table_query"), + "$cleared|$cleared", 'clearing the statistics snapshot exposes the new counters'); +is($reader_query->('SELECT count(*) FROM vm_heap'), '1000', + 'refreshing statistics leaves the MVCC snapshot unchanged'); +$in = "COMMIT;\n\\q\n"; +$reader->finish; +is($node->safe_psql('postgres', $table_query), "$cleared|$cleared", + 'committing the reader adds no clearings'); +is($node->safe_psql('postgres', qq{ +SELECT frozen_page_marks_cleared - $db_frozen, + visible_page_marks_cleared - $db_visible +FROM pg_stat_vacuum_database WHERE datname = current_database() +}), $cleared . '|' . ($cleared + $visible_pages), + 'database totals accumulate both rounds of heap clearings'); + +my $db_totals = $node->safe_psql('postgres', $db_query); +$node->restart; +is($node->safe_psql('postgres', $table_query), "$cleared|$cleared", + 'table VM counters survive a clean restart'); +is($node->safe_psql('postgres', $visible_query), "0|$visible_pages", + 'counts from aborted DML survive a clean restart'); +is($node->safe_psql('postgres', $db_query), $db_totals, + 'database VM counters survive a clean restart'); + +$node->safe_psql('postgres', + q{SELECT pg_stat_reset_single_table_counters('vm_heap'::regclass)}); +wait_for_stats(q{ +SELECT frozen_page_marks_cleared = 0 AND visible_page_marks_cleared = 0 +FROM pg_stat_vacuum_tables WHERE relname = 'vm_heap' +}, 'relation reset'); +is($node->safe_psql('postgres', $table_query), '0|0', + 'relation reset clears both VM counters'); +is($node->safe_psql('postgres', $visible_query), "0|$visible_pages", + 'relation reset preserves another table\'s counters'); +is($node->safe_psql('postgres', $db_query), $db_totals, + 'relation reset preserves database totals'); + +$node->safe_psql('postgres', 'SELECT pg_stat_reset()'); +wait_for_stats(q{ +SELECT frozen_page_marks_cleared = 0 AND visible_page_marks_cleared = 0 +FROM pg_stat_vacuum_database WHERE datname = current_database() +}, 'database reset'); +is($node->safe_psql('postgres', $db_query), '0|0', + 'database reset clears both VM counters'); +is($node->safe_psql('postgres', $visible_query), '0|0', + 'database reset also clears relation VM counters'); + +# The SQL interface belongs to the extension, not the system catalog. +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM pg_attribute +WHERE attrelid IN ('pg_stat_all_tables'::regclass, 'pg_stat_database'::regclass) + AND attname IN ('frozen_page_marks_cleared', 'visible_page_marks_cleared') + AND NOT attisdropped +}), '0', 'core statistics views retain their original columns'); + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/t/005_vacuum_interrupts.pl b/contrib/vacuum_stats/t/005_vacuum_interrupts.pl new file mode 100644 index 00000000000..57967fc491a --- /dev/null +++ b/contrib/vacuum_stats/t/005_vacuum_interrupts.pl @@ -0,0 +1,133 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Adapt v44-0004 to the PG14 collector: wait for heap processing before +# canceling VACUUM, then wait for its deferred statistics report. +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('vacuum_interrupts'); +$node->init; +$node->append_conf('postgresql.conf', "autovacuum = off\n"); +$node->start; +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats'); + +my $db_count = q{SELECT vacuum_interrupt_count FROM pg_stat_vacuum_database + WHERE datname = current_database()}; +my $shared_count = q{SELECT vacuum_interrupt_count FROM pg_stat_vacuum_database + WHERE datid = 0}; +my $nrows = 1000 * ($node->safe_psql('postgres', 'SHOW block_size') / 8192); +$node->safe_psql('postgres', qq{ +CREATE TABLE vacstat_int (id int PRIMARY KEY) + WITH (autovacuum_enabled = off, fillfactor = 10); +INSERT INTO vacstat_int SELECT generate_series(1, $nrows); +DELETE FROM vacstat_int WHERE id % 2 = 0; +}); + +# VERBOSE emits lower-severity messages while the vacuum error callback is +# installed. They must not be mistaken for an interrupted vacuum. +$node->safe_psql('postgres', 'VACUUM (VERBOSE, INDEX_CLEANUP ON) vacstat_int'); +$node->poll_query_until('postgres', q{ +SELECT vacuum_count = 1 FROM pg_stat_all_tables_internal WHERE relname = 'vacstat_int' +}) or BAIL_OUT('successful vacuum report did not reach the collector'); +is($node->safe_psql('postgres', $db_count), '0', + 'successful vacuum and its VERBOSE messages do not count as errors'); + +my ($stdout, $stderr) = ('', ''); +is($node->psql('postgres', 'SELECT 1 / 0', + stdout => \$stdout, stderr => \$stderr), 3, + 'unrelated statement raises an error'); +is($node->safe_psql('postgres', $db_count), '0', + 'an error outside VACUUM does not increment the counter'); + +sub cancel_vacuum +{ + my ($relation, $track_counts) = @_; + my ($in, $out) = ('', ''); + my $timer = IPC::Run::timeout($TestLib::timeout_default); + my $vac = $node->background_psql('postgres', \$in, \$out, $timer, + on_error_stop => 0); + $out = ''; + $in = qq{ +SET application_name = 'vacuum_interrupt_test'; +SET track_counts = $track_counts; +SET vacuum_cost_delay = '100ms'; +SET vacuum_cost_limit = 1; +\\echo vacuum_started +VACUUM (DISABLE_PAGE_SKIPPING) $relation; +\\echo vacuum_done :ERROR :SQLSTATE +}; + pump_until($vac, $timer, \$out, qr/^vacuum_started\r?$/m) + or BAIL_OUT('background psql did not launch VACUUM'); + + # An active VACUUM query alone is insufficient: it might still be waiting + # for a relation lock, before the heap error callback has been installed. + $node->poll_query_until('postgres', qq{ +SELECT count(*) = 1 +FROM pg_stat_activity a JOIN pg_stat_progress_vacuum v USING (pid) +WHERE a.application_name = 'vacuum_interrupt_test' + AND v.relid = '$relation'::regclass AND v.phase = 'scanning heap' + AND a.wait_event = 'VacuumDelay' +}) or BAIL_OUT("VACUUM of $relation did not enter heap processing"); + is($node->safe_psql('postgres', q{ +SELECT pg_cancel_backend(pid) FROM pg_stat_activity +WHERE application_name = 'vacuum_interrupt_test' +}), 't', "sent cancellation to VACUUM of $relation"); + pump_until($vac, $timer, \$out, qr/^vacuum_done\b/m) + or BAIL_OUT('canceled VACUUM did not return'); + like($out, qr/^vacuum_done true 57014\r?$/m, + "VACUUM of $relation reports query_canceled"); + $in = "\\q\n"; + $vac->finish; +} + +for my $expected (1..2) +{ + cancel_vacuum('vacstat_int', 'on'); + ok($node->poll_query_until('postgres', + "SELECT ($db_count) = $expected"), + "canceled vacuum increments database total to exactly $expected"); +} +is($node->safe_psql('postgres', $shared_count), '0', + 'local relation errors leave shared-object statistics alone'); + +cancel_vacuum('vacstat_int', 'off'); +# Drain reports, including the canceled backend's final message, before +# asserting that a disabled counter did not change. +$node->restart; +is($node->safe_psql('postgres', $db_count), '2', + 'track_counts off suppresses counting; earlier errors survive restart'); + +# Make pg_authid large enough to observe cost-delay waits even with 32kB +# blocks. Shared relations report errors to datid zero, not the session's DB. +$node->safe_psql('postgres', qq{ +DO \$\$ BEGIN + FOR i IN 1..$nrows LOOP + EXECUTE format('CREATE ROLE vacstat_role_%s', i); + END LOOP; +END \$\$; +}); +cancel_vacuum('pg_authid', 'on'); +ok($node->poll_query_until('postgres', "SELECT ($shared_count) = 1"), + 'shared relation error reaches the datid zero entry'); +is($node->safe_psql('postgres', $db_count), '2', + 'shared relation error leaves the current database total unchanged'); + +$node->safe_psql('postgres', 'SELECT pg_stat_reset()'); +ok($node->poll_query_until('postgres', "SELECT ($db_count) = 0"), + 'ordinary database reset clears vacuum interruptions'); +$node->restart; +is($node->safe_psql('postgres', $db_count), '0', + 'reset value survives restart'); +is($node->safe_psql('postgres', $shared_count), '1', + 'database reset preserves shared errors, which survive restart'); +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM pg_attribute +WHERE attrelid = 'pg_stat_vacuum_database'::regclass + AND attname = 'vacuum_interrupt_count' AND NOT attisdropped +}), '1', 'extension view exposes vacuum_interrupt_count'); + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/t/006_extension_catalog.pl b/contrib/vacuum_stats/t/006_extension_catalog.pl new file mode 100644 index 00000000000..900542c17ef --- /dev/null +++ b/contrib/vacuum_stats/t/006_extension_catalog.pl @@ -0,0 +1,90 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# SQL access belongs to an extension, including counters collected while the +# extended vacuum statistics are disabled. Check installation in another schema +# and removal/reinstallation without altering the system statistics views. +use strict; +use warnings FATAL => 'all'; +use PostgresNode; +use TestLib; +use Test::More; + +my $node = get_new_node('extension_catalog'); +$node->init; +$node->append_conf('postgresql.conf', "autovacuum = off\n"); +$node->start; + +my $catalog_query = q{ +SELECT c.relname, pg_get_viewdef(c.oid), a.attnum, a.attname, a.atttypid +FROM pg_class c JOIN pg_namespace n ON n.oid = c.relnamespace +JOIN pg_attribute a ON a.attrelid = c.oid AND a.attnum > 0 +WHERE n.nspname = 'pg_catalog' AND c.relkind = 'v' + AND (c.relname LIKE 'pg_stat_%' OR c.relname LIKE 'gp_stat_%') +ORDER BY c.relname, a.attnum +}; +my $catalog_before = $node->safe_psql('postgres', $catalog_query); +$node->safe_psql('postgres', q{ +CREATE SCHEMA maintenance; +CREATE EXTENSION vacuum_stats SCHEMA maintenance; +}); +is($node->safe_psql('postgres', $catalog_query), $catalog_before, + 'installing the extension preserves system statistics view definitions and columns'); +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM pg_proc p JOIN pg_namespace n ON n.oid = p.pronamespace +WHERE n.nspname = 'pg_catalog' + AND p.proname IN ('pg_stat_get_frozen_page_marks_cleared', + 'pg_stat_get_visible_page_marks_cleared', + 'pg_stat_get_total_vacuum_time', 'pg_stat_get_total_autovacuum_time', + 'pg_stat_get_total_analyze_time', 'pg_stat_get_total_autoanalyze_time', + 'pg_stat_get_total_vacuum_delay_time', 'pg_stat_get_total_autovacuum_delay_time', + 'pg_stat_get_vacuum_failsafe_count', + 'pg_stat_get_db_frozen_page_marks_cleared', + 'pg_stat_get_db_visible_page_marks_cleared', + 'pg_stat_get_db_total_vacuum_time', 'pg_stat_get_db_total_autovacuum_time', + 'pg_stat_get_db_total_vacuum_delay_time', 'pg_stat_get_db_total_autovacuum_delay_time', + 'pg_stat_get_db_vacuum_failsafe_count', 'pg_stat_get_db_vacuum_interrupt_count') +}), '0', 'new statistics getters are not installed as built-in functions'); + +is($node->safe_psql('postgres', 'SHOW track_vacuum_statistics'), 'off', + 'ordinary maintenance timing is tested without extended statistics storage'); +$node->safe_psql('postgres', q{ +CREATE TABLE analyze_time (id int); +INSERT INTO analyze_time SELECT generate_series(1, 10000); +ANALYZE analyze_time; +}); +$node->poll_query_until('postgres', q{ +SELECT total_analyze_time > 0 FROM maintenance.pg_stat_vacuum_tables +WHERE relname = 'analyze_time' +}) or BAIL_OUT('ANALYZE timing report did not reach the collector'); +my $analyze_query = q{ +SELECT total_analyze_time, total_autoanalyze_time +FROM maintenance.pg_stat_vacuum_tables WHERE relname = 'analyze_time' +}; +my $analyze_before = $node->safe_psql('postgres', $analyze_query); +is($node->safe_psql('postgres', q{ +SELECT total_analyze_time > 0 AND total_autoanalyze_time = 0 +FROM maintenance.gp_stat_vacuum_tables WHERE relname = 'analyze_time' +}), 't', 'cluster view resolves extension getters in a non-default schema'); +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM maintenance.pg_stat_vacuum_database +WHERE datid = 0 AND datname IS NULL +}), '1', 'local database view exposes the shared-relation entry'); +is($node->safe_psql('postgres', q{ +SELECT count(*) FROM maintenance.gp_stat_vacuum_database +WHERE datid = 0 AND datname IS NULL +}), '1', 'cluster database view exposes the shared entry once in utility mode'); + +$node->safe_psql('postgres', 'SELECT maintenance.vacuum_stats_reset()'); +# Drain the asynchronous reset before testing fields it must preserve. +$node->restart; +is($node->safe_psql('postgres', $analyze_query), $analyze_before, + 'dedicated vacuum reset preserves ANALYZE times'); +$node->safe_psql('postgres', 'DROP EXTENSION vacuum_stats'); +is($node->safe_psql('postgres', $catalog_query), $catalog_before, + 'dropping the extension preserves system statistics views'); +$node->safe_psql('postgres', 'CREATE EXTENSION vacuum_stats SCHEMA maintenance'); +is($node->safe_psql('postgres', $analyze_query), $analyze_before, + 'reinstalling the extension reads the same collector statistics'); + +$node->stop; +done_testing(); diff --git a/contrib/vacuum_stats/vacuum_stats--1.0.sql b/contrib/vacuum_stats/vacuum_stats--1.0.sql new file mode 100644 index 00000000000..eea69643382 --- /dev/null +++ b/contrib/vacuum_stats/vacuum_stats--1.0.sql @@ -0,0 +1,588 @@ +/* contrib/vacuum_stats/vacuum_stats--1.0.sql */ + +-- complain if script is sourced in psql, rather than via CREATE EXTENSION +\echo Use "CREATE EXTENSION vacuum_stats" to load this file. \quit + +-- +-- Per-relation accessor functions (tables and indexes). +-- +CREATE FUNCTION pg_stat_get_vacuum_tuples_deleted(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_tuples_deleted' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_dead_tuples(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_dead_tuples' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_pages_deleted(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_pages_deleted' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_bytes_removed(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_bytes_removed' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_total_file_segs(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_total_file_segs' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_dead_pages(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_dead_pages' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_pages_frozen(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_pages_frozen' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_pages_all_visible(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_pages_all_visible' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_vacuum_freeze_age_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_freeze_age_count' +LANGUAGE C STABLE STRICT; + +-- +-- Per-database accessor functions. +-- +CREATE FUNCTION pg_stat_get_db_vacuum_tuples_deleted(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_tuples_deleted' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_dead_tuples(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_dead_tuples' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_pages_deleted(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_pages_deleted' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_bytes_removed(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_bytes_removed' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_dead_pages(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_dead_pages' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_pages_frozen(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_pages_frozen' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_pages_all_visible(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_pages_all_visible' +LANGUAGE C STABLE STRICT; + +CREATE FUNCTION pg_stat_get_db_vacuum_freeze_age_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_freeze_age_count' +LANGUAGE C STABLE STRICT; + +-- Access to counters stored in ordinary relation and database statistics. +CREATE FUNCTION pg_stat_get_frozen_page_marks_cleared(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_frozen_page_marks_cleared' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_visible_page_marks_cleared(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_visible_page_marks_cleared' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_vacuum_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_vacuum_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_autovacuum_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_autovacuum_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_vacuum_delay_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_vacuum_delay_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_autovacuum_delay_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_autovacuum_delay_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_analyze_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_analyze_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_total_autoanalyze_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_total_autoanalyze_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_vacuum_failsafe_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_vacuum_failsafe_count' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_frozen_page_marks_cleared(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_frozen_page_marks_cleared' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_visible_page_marks_cleared(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_visible_page_marks_cleared' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_total_vacuum_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_vacuum_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_total_autovacuum_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_autovacuum_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_total_vacuum_delay_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_vacuum_delay_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_total_autovacuum_delay_time(oid) RETURNS double precision +AS 'MODULE_PATHNAME', 'pg_stat_get_db_total_autovacuum_delay_time' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_vacuum_failsafe_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_failsafe_count' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +CREATE FUNCTION pg_stat_get_db_vacuum_interrupt_count(oid) RETURNS int8 +AS 'MODULE_PATHNAME', 'pg_stat_get_db_vacuum_interrupt_count' +LANGUAGE C STABLE STRICT PARALLEL RESTRICTED; + +-- Local statistics, in milliseconds for timing columns. +CREATE VIEW pg_stat_vacuum_tables AS + SELECT + c.oid AS relid, + n.nspname AS schemaname, + c.relname AS relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(c.oid) AS tuples_deleted, + @extschema@.pg_stat_get_vacuum_dead_tuples(c.oid) AS dead_tuples, + @extschema@.pg_stat_get_vacuum_pages_deleted(c.oid) AS pages_deleted, + @extschema@.pg_stat_get_vacuum_bytes_removed(c.oid) AS bytes_removed, + @extschema@.pg_stat_get_vacuum_dead_pages(c.oid) AS dead_pages, + @extschema@.pg_stat_get_vacuum_pages_frozen(c.oid) AS pages_frozen, + @extschema@.pg_stat_get_vacuum_pages_all_visible(c.oid) AS pages_all_visible, + @extschema@.pg_stat_get_frozen_page_marks_cleared(c.oid) AS frozen_page_marks_cleared, + @extschema@.pg_stat_get_visible_page_marks_cleared(c.oid) AS visible_page_marks_cleared, + @extschema@.pg_stat_get_vacuum_freeze_age_count(c.oid) AS freeze_age_vacuum_count, + @extschema@.pg_stat_get_vacuum_failsafe_count(c.oid) AS vacuum_failsafe_count, + @extschema@.pg_stat_get_total_vacuum_time(c.oid) AS total_vacuum_time, + @extschema@.pg_stat_get_total_autovacuum_time(c.oid) AS total_autovacuum_time, + @extschema@.pg_stat_get_total_vacuum_delay_time(c.oid) AS total_vacuum_delay_time, + @extschema@.pg_stat_get_total_autovacuum_delay_time(c.oid) AS total_autovacuum_delay_time, + @extschema@.pg_stat_get_total_analyze_time(c.oid) AS total_analyze_time, + @extschema@.pg_stat_get_total_autoanalyze_time(c.oid) AS total_autoanalyze_time, + @extschema@.pg_stat_get_vacuum_total_file_segs(c.oid) AS total_file_segs + FROM pg_catalog.pg_class c + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm'); + +CREATE VIEW pg_stat_vacuum_indexes AS + SELECT + c.oid AS relid, + i.oid AS indexrelid, + n.nspname AS schemaname, + c.relname AS relname, + i.relname AS indexrelname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(i.oid) AS tuples_deleted, + @extschema@.pg_stat_get_vacuum_dead_tuples(i.oid) AS dead_tuples, + @extschema@.pg_stat_get_vacuum_pages_deleted(i.oid) AS pages_deleted, + @extschema@.pg_stat_get_vacuum_bytes_removed(i.oid) AS bytes_removed, + @extschema@.pg_stat_get_vacuum_dead_pages(i.oid) AS dead_pages, + @extschema@.pg_stat_get_vacuum_pages_frozen(i.oid) AS pages_frozen, + @extschema@.pg_stat_get_vacuum_pages_all_visible(i.oid) AS pages_all_visible, + @extschema@.pg_stat_get_frozen_page_marks_cleared(i.oid) AS frozen_page_marks_cleared, + @extschema@.pg_stat_get_visible_page_marks_cleared(i.oid) AS visible_page_marks_cleared, + @extschema@.pg_stat_get_vacuum_freeze_age_count(i.oid) AS freeze_age_vacuum_count, + @extschema@.pg_stat_get_vacuum_failsafe_count(i.oid) AS vacuum_failsafe_count, + @extschema@.pg_stat_get_total_vacuum_time(i.oid) AS total_vacuum_time, + @extschema@.pg_stat_get_total_autovacuum_time(i.oid) AS total_autovacuum_time, + @extschema@.pg_stat_get_total_vacuum_delay_time(i.oid) AS total_vacuum_delay_time, + @extschema@.pg_stat_get_total_autovacuum_delay_time(i.oid) AS total_autovacuum_delay_time + FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_index x ON c.oid = x.indrelid + JOIN pg_catalog.pg_class i ON i.oid = x.indexrelid + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') AND i.relkind = 'i'; + +CREATE VIEW pg_stat_vacuum_database AS + SELECT + d.oid AS datid, + d.datname AS datname, + @extschema@.pg_stat_get_db_vacuum_tuples_deleted(d.oid) AS tuples_deleted, + @extschema@.pg_stat_get_db_vacuum_dead_tuples(d.oid) AS dead_tuples, + @extschema@.pg_stat_get_db_vacuum_pages_deleted(d.oid) AS pages_deleted, + @extschema@.pg_stat_get_db_vacuum_bytes_removed(d.oid) AS bytes_removed, + @extschema@.pg_stat_get_db_vacuum_dead_pages(d.oid) AS dead_pages, + @extschema@.pg_stat_get_db_vacuum_pages_frozen(d.oid) AS pages_frozen, + @extschema@.pg_stat_get_db_vacuum_pages_all_visible(d.oid) AS pages_all_visible, + @extschema@.pg_stat_get_db_frozen_page_marks_cleared(d.oid) AS frozen_page_marks_cleared, + @extschema@.pg_stat_get_db_visible_page_marks_cleared(d.oid) AS visible_page_marks_cleared, + @extschema@.pg_stat_get_db_vacuum_freeze_age_count(d.oid) AS freeze_age_vacuum_count, + @extschema@.pg_stat_get_db_vacuum_failsafe_count(d.oid) AS vacuum_failsafe_count, + @extschema@.pg_stat_get_db_total_vacuum_time(d.oid) AS total_vacuum_time, + @extschema@.pg_stat_get_db_total_autovacuum_time(d.oid) AS total_autovacuum_time, + @extschema@.pg_stat_get_db_total_vacuum_delay_time(d.oid) AS total_vacuum_delay_time, + @extschema@.pg_stat_get_db_total_autovacuum_delay_time(d.oid) AS total_autovacuum_delay_time, + @extschema@.pg_stat_get_db_vacuum_interrupt_count(d.oid) AS vacuum_interrupt_count + FROM ( + SELECT 0::oid AS oid, NULL::name AS datname + UNION ALL + SELECT oid, datname FROM pg_catalog.pg_database + ) d; + +-- Cluster views execute on the coordinator and all segments. The bodies +-- access catalogs directly because segment functions cannot scan local views. +-- Utility sessions return their local rows once, without a segment branch. +CREATE FUNCTION gp_stat_get_coordinator_vacuum_tables() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + c.oid, + n.nspname, + c.relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(c.oid), + @extschema@.pg_stat_get_vacuum_dead_tuples(c.oid), + @extschema@.pg_stat_get_vacuum_pages_deleted(c.oid), + @extschema@.pg_stat_get_vacuum_bytes_removed(c.oid), + @extschema@.pg_stat_get_vacuum_dead_pages(c.oid), + @extschema@.pg_stat_get_vacuum_pages_frozen(c.oid), + @extschema@.pg_stat_get_vacuum_pages_all_visible(c.oid), + @extschema@.pg_stat_get_frozen_page_marks_cleared(c.oid), + @extschema@.pg_stat_get_visible_page_marks_cleared(c.oid), + @extschema@.pg_stat_get_vacuum_freeze_age_count(c.oid), + @extschema@.pg_stat_get_vacuum_failsafe_count(c.oid), + @extschema@.pg_stat_get_total_vacuum_time(c.oid), + @extschema@.pg_stat_get_total_autovacuum_time(c.oid), + @extschema@.pg_stat_get_total_vacuum_delay_time(c.oid), + @extschema@.pg_stat_get_total_autovacuum_delay_time(c.oid), + @extschema@.pg_stat_get_total_analyze_time(c.oid), + @extschema@.pg_stat_get_total_autoanalyze_time(c.oid), + @extschema@.pg_stat_get_vacuum_total_file_segs(c.oid) + FROM pg_catalog.pg_class c + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') +$$ +LANGUAGE SQL EXECUTE ON COORDINATOR; + +CREATE FUNCTION gp_stat_get_segment_vacuum_tables() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + c.oid, + n.nspname, + c.relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(c.oid), + @extschema@.pg_stat_get_vacuum_dead_tuples(c.oid), + @extschema@.pg_stat_get_vacuum_pages_deleted(c.oid), + @extschema@.pg_stat_get_vacuum_bytes_removed(c.oid), + @extschema@.pg_stat_get_vacuum_dead_pages(c.oid), + @extschema@.pg_stat_get_vacuum_pages_frozen(c.oid), + @extschema@.pg_stat_get_vacuum_pages_all_visible(c.oid), + @extschema@.pg_stat_get_frozen_page_marks_cleared(c.oid), + @extschema@.pg_stat_get_visible_page_marks_cleared(c.oid), + @extschema@.pg_stat_get_vacuum_freeze_age_count(c.oid), + @extschema@.pg_stat_get_vacuum_failsafe_count(c.oid), + @extschema@.pg_stat_get_total_vacuum_time(c.oid), + @extschema@.pg_stat_get_total_autovacuum_time(c.oid), + @extschema@.pg_stat_get_total_vacuum_delay_time(c.oid), + @extschema@.pg_stat_get_total_autovacuum_delay_time(c.oid), + @extschema@.pg_stat_get_total_analyze_time(c.oid), + @extschema@.pg_stat_get_total_autoanalyze_time(c.oid), + @extschema@.pg_stat_get_vacuum_total_file_segs(c.oid) + FROM pg_catalog.pg_class c + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') + AND pg_catalog.current_setting('gp_role') <> 'utility' +$$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +CREATE VIEW gp_stat_vacuum_tables AS + SELECT * FROM gp_stat_get_coordinator_vacuum_tables() AS T + (gp_segment_id int, + relid oid, + schemaname name, + relname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision, + total_analyze_time double precision, + total_autoanalyze_time double precision, + total_file_segs int8) + UNION ALL + SELECT * FROM gp_stat_get_segment_vacuum_tables() AS T + (gp_segment_id int, + relid oid, + schemaname name, + relname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision, + total_analyze_time double precision, + total_autoanalyze_time double precision, + total_file_segs int8); + +CREATE FUNCTION gp_stat_get_coordinator_vacuum_indexes() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + c.oid, + i.oid, + n.nspname, + c.relname, + i.relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(i.oid), + @extschema@.pg_stat_get_vacuum_dead_tuples(i.oid), + @extschema@.pg_stat_get_vacuum_pages_deleted(i.oid), + @extschema@.pg_stat_get_vacuum_bytes_removed(i.oid), + @extschema@.pg_stat_get_vacuum_dead_pages(i.oid), + @extschema@.pg_stat_get_vacuum_pages_frozen(i.oid), + @extschema@.pg_stat_get_vacuum_pages_all_visible(i.oid), + @extschema@.pg_stat_get_frozen_page_marks_cleared(i.oid), + @extschema@.pg_stat_get_visible_page_marks_cleared(i.oid), + @extschema@.pg_stat_get_vacuum_freeze_age_count(i.oid), + @extschema@.pg_stat_get_vacuum_failsafe_count(i.oid), + @extschema@.pg_stat_get_total_vacuum_time(i.oid), + @extschema@.pg_stat_get_total_autovacuum_time(i.oid), + @extschema@.pg_stat_get_total_vacuum_delay_time(i.oid), + @extschema@.pg_stat_get_total_autovacuum_delay_time(i.oid) + FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_index x ON c.oid = x.indrelid + JOIN pg_catalog.pg_class i ON i.oid = x.indexrelid + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') AND i.relkind = 'i' +$$ +LANGUAGE SQL EXECUTE ON COORDINATOR; + +CREATE FUNCTION gp_stat_get_segment_vacuum_indexes() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + c.oid, + i.oid, + n.nspname, + c.relname, + i.relname, + @extschema@.pg_stat_get_vacuum_tuples_deleted(i.oid), + @extschema@.pg_stat_get_vacuum_dead_tuples(i.oid), + @extschema@.pg_stat_get_vacuum_pages_deleted(i.oid), + @extschema@.pg_stat_get_vacuum_bytes_removed(i.oid), + @extschema@.pg_stat_get_vacuum_dead_pages(i.oid), + @extschema@.pg_stat_get_vacuum_pages_frozen(i.oid), + @extschema@.pg_stat_get_vacuum_pages_all_visible(i.oid), + @extschema@.pg_stat_get_frozen_page_marks_cleared(i.oid), + @extschema@.pg_stat_get_visible_page_marks_cleared(i.oid), + @extschema@.pg_stat_get_vacuum_freeze_age_count(i.oid), + @extschema@.pg_stat_get_vacuum_failsafe_count(i.oid), + @extschema@.pg_stat_get_total_vacuum_time(i.oid), + @extschema@.pg_stat_get_total_autovacuum_time(i.oid), + @extschema@.pg_stat_get_total_vacuum_delay_time(i.oid), + @extschema@.pg_stat_get_total_autovacuum_delay_time(i.oid) + FROM pg_catalog.pg_class c + JOIN pg_catalog.pg_index x ON c.oid = x.indrelid + JOIN pg_catalog.pg_class i ON i.oid = x.indexrelid + LEFT JOIN pg_catalog.pg_namespace n ON n.oid = c.relnamespace + WHERE c.relkind IN ('r', 't', 'm') AND i.relkind = 'i' + AND pg_catalog.current_setting('gp_role') <> 'utility' +$$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +CREATE VIEW gp_stat_vacuum_indexes AS + SELECT * FROM gp_stat_get_coordinator_vacuum_indexes() AS T + (gp_segment_id int, + relid oid, + indexrelid oid, + schemaname name, + relname name, + indexrelname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision) + UNION ALL + SELECT * FROM gp_stat_get_segment_vacuum_indexes() AS T + (gp_segment_id int, + relid oid, + indexrelid oid, + schemaname name, + relname name, + indexrelname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision); + +CREATE FUNCTION gp_stat_get_coordinator_vacuum_database() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + d.oid, + d.datname, + @extschema@.pg_stat_get_db_vacuum_tuples_deleted(d.oid), + @extschema@.pg_stat_get_db_vacuum_dead_tuples(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_deleted(d.oid), + @extschema@.pg_stat_get_db_vacuum_bytes_removed(d.oid), + @extschema@.pg_stat_get_db_vacuum_dead_pages(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_frozen(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_all_visible(d.oid), + @extschema@.pg_stat_get_db_frozen_page_marks_cleared(d.oid), + @extschema@.pg_stat_get_db_visible_page_marks_cleared(d.oid), + @extschema@.pg_stat_get_db_vacuum_freeze_age_count(d.oid), + @extschema@.pg_stat_get_db_vacuum_failsafe_count(d.oid), + @extschema@.pg_stat_get_db_total_vacuum_time(d.oid), + @extschema@.pg_stat_get_db_total_autovacuum_time(d.oid), + @extschema@.pg_stat_get_db_total_vacuum_delay_time(d.oid), + @extschema@.pg_stat_get_db_total_autovacuum_delay_time(d.oid), + @extschema@.pg_stat_get_db_vacuum_interrupt_count(d.oid) + FROM ( + SELECT 0::oid AS oid, NULL::name AS datname + UNION ALL + SELECT oid, datname FROM pg_catalog.pg_database + ) d +$$ +LANGUAGE SQL EXECUTE ON COORDINATOR; + +CREATE FUNCTION gp_stat_get_segment_vacuum_database() RETURNS SETOF RECORD AS +$$ + SELECT pg_catalog.gp_execution_segment() AS gp_segment_id, + d.oid, + d.datname, + @extschema@.pg_stat_get_db_vacuum_tuples_deleted(d.oid), + @extschema@.pg_stat_get_db_vacuum_dead_tuples(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_deleted(d.oid), + @extschema@.pg_stat_get_db_vacuum_bytes_removed(d.oid), + @extschema@.pg_stat_get_db_vacuum_dead_pages(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_frozen(d.oid), + @extschema@.pg_stat_get_db_vacuum_pages_all_visible(d.oid), + @extschema@.pg_stat_get_db_frozen_page_marks_cleared(d.oid), + @extschema@.pg_stat_get_db_visible_page_marks_cleared(d.oid), + @extschema@.pg_stat_get_db_vacuum_freeze_age_count(d.oid), + @extschema@.pg_stat_get_db_vacuum_failsafe_count(d.oid), + @extschema@.pg_stat_get_db_total_vacuum_time(d.oid), + @extschema@.pg_stat_get_db_total_autovacuum_time(d.oid), + @extschema@.pg_stat_get_db_total_vacuum_delay_time(d.oid), + @extschema@.pg_stat_get_db_total_autovacuum_delay_time(d.oid), + @extschema@.pg_stat_get_db_vacuum_interrupt_count(d.oid) + FROM ( + SELECT 0::oid AS oid, NULL::name AS datname + UNION ALL + SELECT oid, datname FROM pg_catalog.pg_database + ) d + WHERE pg_catalog.current_setting('gp_role') <> 'utility' +$$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +CREATE VIEW gp_stat_vacuum_database AS + SELECT * FROM gp_stat_get_coordinator_vacuum_database() AS T + (gp_segment_id int, + datid oid, + datname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision, + vacuum_interrupt_count int8) + UNION ALL + SELECT * FROM gp_stat_get_segment_vacuum_database() AS T + (gp_segment_id int, + datid oid, + datname name, + tuples_deleted int8, + dead_tuples int8, + pages_deleted int8, + bytes_removed int8, + dead_pages int8, + pages_frozen int8, + pages_all_visible int8, + frozen_page_marks_cleared int8, + visible_page_marks_cleared int8, + freeze_age_vacuum_count int8, + vacuum_failsafe_count int8, + total_vacuum_time double precision, + total_autovacuum_time double precision, + total_vacuum_delay_time double precision, + total_autovacuum_delay_time double precision, + vacuum_interrupt_count int8); + +-- +-- Resetting, for when only these counters are in the way: pg_stat_reset() +-- and pg_stat_reset_single_table_counters() throw away the rest of the +-- statistics of the database or the relation as well. +-- +-- As with the rest of the statistics, resetting acts on the node it runs on, +-- so on a cluster both the coordinator function and the segment one have to +-- be called. +-- +CREATE FUNCTION vacuum_stats_reset() RETURNS void +AS 'MODULE_PATHNAME', 'vacuum_stats_reset' +LANGUAGE C; + +CREATE FUNCTION vacuum_stats_reset(relid oid) RETURNS void +AS 'MODULE_PATHNAME', 'vacuum_stats_reset_relation' +LANGUAGE C STRICT; + +CREATE FUNCTION gp_vacuum_stats_reset() RETURNS SETOF void AS +$$ SELECT @extschema@.vacuum_stats_reset() $$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +CREATE FUNCTION gp_vacuum_stats_reset(relid oid) RETURNS SETOF void AS +$$ SELECT @extschema@.vacuum_stats_reset($1) $$ +LANGUAGE SQL EXECUTE ON ALL SEGMENTS; + +REVOKE ALL ON FUNCTION vacuum_stats_reset() FROM PUBLIC; +REVOKE ALL ON FUNCTION vacuum_stats_reset(oid) FROM PUBLIC; +REVOKE ALL ON FUNCTION gp_vacuum_stats_reset() FROM PUBLIC; +REVOKE ALL ON FUNCTION gp_vacuum_stats_reset(oid) FROM PUBLIC; + +GRANT SELECT ON pg_stat_vacuum_tables, pg_stat_vacuum_indexes, + pg_stat_vacuum_database, gp_stat_vacuum_tables, + gp_stat_vacuum_indexes, gp_stat_vacuum_database TO PUBLIC; diff --git a/contrib/vacuum_stats/vacuum_stats.c b/contrib/vacuum_stats/vacuum_stats.c new file mode 100644 index 00000000000..161881a4e97 --- /dev/null +++ b/contrib/vacuum_stats/vacuum_stats.c @@ -0,0 +1,189 @@ +/*------------------------------------------------------------------------- + * + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + * + * vacuum_stats.c + * Expose the vacuum counters accumulated by the statistics collector + * for relations (tables and indexes) and databases. + * + * The backend collects these counters in ordinary statistics entries and, + * when enabled, extended vacuum statistics. SQL access belongs to this + * extension; read both groups through the regular pgstat fetch API. + * + * contrib/vacuum_stats/vacuum_stats.c + * + *------------------------------------------------------------------------- + */ +#include "postgres.h" + +#include "fmgr.h" +#include "pgstat.h" + +PG_MODULE_MAGIC; + +/* Ordinary counters follow track_counts, independently of extended tracking. */ +#define DEFINE_REL_COUNTER_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + PgStat_StatTabEntry *entry = pgstat_fetch_stat_tabentry(PG_GETARG_OID(0)); \ + PG_RETURN_INT64(entry ? (int64) entry->field : 0); \ +} + +#define DEFINE_DB_COUNTER_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + PgStat_StatDBEntry *entry = pgstat_fetch_stat_dbentry(PG_GETARG_OID(0)); \ + PG_RETURN_INT64(entry ? (int64) entry->field : 0); \ +} + +/* Store microseconds in the collector and expose milliseconds without rounding. */ +#define DEFINE_REL_TIME_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + PgStat_StatTabEntry *entry = pgstat_fetch_stat_tabentry(PG_GETARG_OID(0)); \ + PG_RETURN_FLOAT8(entry ? (double) entry->field / 1000.0 : 0); \ +} + +#define DEFINE_DB_TIME_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + PgStat_StatDBEntry *entry = pgstat_fetch_stat_dbentry(PG_GETARG_OID(0)); \ + PG_RETURN_FLOAT8(entry ? (double) entry->field / 1000.0 : 0); \ +} + +DEFINE_REL_COUNTER_FUNC(pg_stat_get_frozen_page_marks_cleared, frozen_page_marks_cleared) +DEFINE_REL_COUNTER_FUNC(pg_stat_get_visible_page_marks_cleared, visible_page_marks_cleared) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_vacuum_time, total_vacuum_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_autovacuum_time, total_autovacuum_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_vacuum_delay_time, total_vacuum_delay_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_autovacuum_delay_time, total_autovacuum_delay_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_analyze_time, total_analyze_time) +DEFINE_REL_TIME_FUNC(pg_stat_get_total_autoanalyze_time, total_autoanalyze_time) +DEFINE_REL_COUNTER_FUNC(pg_stat_get_vacuum_failsafe_count, vacuum_failsafe_count) +DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_frozen_page_marks_cleared, n_frozen_page_marks_cleared) +DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_visible_page_marks_cleared, n_visible_page_marks_cleared) +DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_vacuum_time, total_vacuum_time) +DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_autovacuum_time, total_autovacuum_time) +DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_vacuum_delay_time, total_vacuum_delay_time) +DEFINE_DB_TIME_FUNC(pg_stat_get_db_total_autovacuum_delay_time, total_autovacuum_delay_time) +DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_vacuum_failsafe_count, vacuum_failsafe_count) +DEFINE_DB_COUNTER_FUNC(pg_stat_get_db_vacuum_interrupt_count, vacuum_interrupt_count) + +/* Fetch the relation's vacuum counters, or NULL when unavailable. */ +static PgStat_VacuumStats * +fetch_rel_vacuum_stats(Oid relid) +{ + return pgstat_fetch_stat_vacuum_stats(relid); +} + +/* + * Fetch the per-database vacuum counters, or NULL if the statistics + * collector has no entry for the database. + */ +static PgStat_VacuumStats * +fetch_db_vacuum_stats(Oid dbid) +{ + PgStat_StatDBEntry *dbentry; + + if (!pgstat_track_vacuum_statistics) + return NULL; + + dbentry = pgstat_fetch_stat_dbentry(dbid); + if (dbentry == NULL) + return NULL; + + return &dbentry->n_vacuum_stats; +} + +#define DEFINE_REL_VACSTAT_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + Oid relid = PG_GETARG_OID(0); \ + PgStat_VacuumStats *stats = fetch_rel_vacuum_stats(relid); \ +\ + PG_RETURN_INT64(stats ? (int64) stats->field : 0); \ +} + +#define DEFINE_DB_VACSTAT_FUNC(funcname, field) \ +PG_FUNCTION_INFO_V1(funcname); \ +Datum \ +funcname(PG_FUNCTION_ARGS) \ +{ \ + Oid dbid = PG_GETARG_OID(0); \ + PgStat_VacuumStats *stats = fetch_db_vacuum_stats(dbid); \ +\ + PG_RETURN_INT64(stats ? (int64) stats->field : 0); \ +} + +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_tuples_deleted, tuples_deleted) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_dead_tuples, dead_tuples) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_pages_deleted, pages_deleted) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_bytes_removed, bytes_removed) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_total_file_segs, total_file_segs) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_dead_pages, dead_pages) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_pages_frozen, pages_frozen) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_pages_all_visible, pages_all_visible) +DEFINE_REL_VACSTAT_FUNC(pg_stat_get_vacuum_freeze_age_count, freeze_age_vacuum_count) + +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_tuples_deleted, tuples_deleted) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_dead_tuples, dead_tuples) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_deleted, pages_deleted) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_bytes_removed, bytes_removed) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_dead_pages, dead_pages) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_frozen, pages_frozen) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_pages_all_visible, pages_all_visible) +DEFINE_DB_VACSTAT_FUNC(pg_stat_get_db_vacuum_freeze_age_count, freeze_age_vacuum_count) + +/* + * Throw away the vacuum counters of one relation, or of the whole database, + * without touching the rest of the statistics -- which is what + * pg_stat_reset() and pg_stat_reset_single_table_counters() would do. + * + * Like the other resetting functions this acts on the node it runs on, so on + * a cluster it has to be dispatched to the segments as well; see + * gp_vacuum_stats_reset() in the extension script. + */ +PG_FUNCTION_INFO_V1(vacuum_stats_reset); +Datum +vacuum_stats_reset(PG_FUNCTION_ARGS) +{ + pgstat_reset_vacuum_stats(InvalidOid, true); + + PG_RETURN_VOID(); +} + +PG_FUNCTION_INFO_V1(vacuum_stats_reset_relation); +Datum +vacuum_stats_reset_relation(PG_FUNCTION_ARGS) +{ + Oid relid = PG_GETARG_OID(0); + + pgstat_reset_vacuum_stats(relid, false); + + PG_RETURN_VOID(); +} diff --git a/contrib/vacuum_stats/vacuum_stats.control b/contrib/vacuum_stats/vacuum_stats.control new file mode 100644 index 00000000000..5f80449984d --- /dev/null +++ b/contrib/vacuum_stats/vacuum_stats.control @@ -0,0 +1,5 @@ +# vacuum_stats extension +comment = 'per-relation and per-database vacuum statistics' +default_version = '1.0' +module_pathname = '$libdir/vacuum_stats' +relocatable = false diff --git a/doc/src/sgml/config.sgml b/doc/src/sgml/config.sgml index 0396704f3d8..ef4e256834b 100644 --- a/doc/src/sgml/config.sgml +++ b/doc/src/sgml/config.sgml @@ -7601,6 +7601,26 @@ COPY postgres_log FROM '/full/path/to/logfile.csv' WITH csv; + + track_cost_delay_timing (boolean) + + track_cost_delay_timing configuration parameter + + + + + Enables timing of cost-based vacuum delay (see + ). This parameter + is off by default, as it will repeatedly query the operating system for + the current time, which may cause significant overhead on some + platforms. You can use the tool to + measure the overhead of timing on your system. The measured time is + included in cumulative vacuum statistics. Only superusers can change + this setting. + + + + track_io_timing (boolean) diff --git a/src/backend/access/aocs/aocs_compaction.c b/src/backend/access/aocs/aocs_compaction.c index e6ecea73de1..41919408c01 100644 --- a/src/backend/access/aocs/aocs_compaction.c +++ b/src/backend/access/aocs/aocs_compaction.c @@ -323,7 +323,7 @@ AOCSSegmentFileFullCompaction(Relation aorel, tupleCount++; if (VacuumCostActive && tupleCount % tuplePerPage == 0) { - vacuum_delay_point(); + vacuum_delay_point(false); } /* diff --git a/src/backend/access/aocs/aocsam_handler.c b/src/backend/access/aocs/aocsam_handler.c index 4b3cd2a52ef..83bba9e0443 100644 --- a/src/backend/access/aocs/aocsam_handler.c +++ b/src/backend/access/aocs/aocsam_handler.c @@ -1681,7 +1681,7 @@ aoco_acquire_sample_rows(Relation onerel, int elevel, HeapTuple *rows, { aocoscan->targrow = RowSampler_Next(&rs); - vacuum_delay_point(); + vacuum_delay_point(true); if (aocs_get_target_tuple(aocoscan, aocoscan->targrow, slot)) { diff --git a/src/backend/access/appendonly/aomd.c b/src/backend/access/appendonly/aomd.c index 342e7771b7b..36dfa5d71f5 100644 --- a/src/backend/access/appendonly/aomd.c +++ b/src/backend/access/appendonly/aomd.c @@ -221,7 +221,8 @@ TruncateAOSegmentFile(File fd, Relation rel, int32 segFileNum, int64 offset, AOV /* report heap-equivalent blocks vacuumed */ vacrelstats->nbytes_truncated += filesize_before - offset; pgstat_progress_update_param(PROGRESS_VACUUM_HEAP_BLKS_VACUUMED, - RelationGuessNumberOfBlocksFromSize(vacrelstats->nbytes_truncated)); + vacrelstats->nbytes_truncated / BLCKSZ + + (vacrelstats->nbytes_truncated % BLCKSZ != 0)); } if (XLogIsNeeded() && RelationNeedsWAL(rel)) diff --git a/src/backend/access/appendonly/appendonly_compaction.c b/src/backend/access/appendonly/appendonly_compaction.c index 05b2143b246..5ad525883b4 100644 --- a/src/backend/access/appendonly/appendonly_compaction.c +++ b/src/backend/access/appendonly/appendonly_compaction.c @@ -506,7 +506,7 @@ AppendOnlySegmentFileFullCompaction(Relation aorel, tupleCount++; if (VacuumCostActive && tupleCount % tuplePerPage == 0) { - vacuum_delay_point(); + vacuum_delay_point(false); } } diff --git a/src/backend/access/appendonly/appendonlyam_handler.c b/src/backend/access/appendonly/appendonlyam_handler.c index 715cebd7579..ad299fecb51 100644 --- a/src/backend/access/appendonly/appendonlyam_handler.c +++ b/src/backend/access/appendonly/appendonlyam_handler.c @@ -1563,7 +1563,7 @@ appendonly_acquire_sample_rows(Relation onerel, int elevel, HeapTuple *rows, { aoscan->targrow = RowSampler_Next(&rs); - vacuum_delay_point(); + vacuum_delay_point(true); if (appendonly_get_target_tuple(aoscan, aoscan->targrow, slot)) { diff --git a/src/backend/access/gin/ginfast.c b/src/backend/access/gin/ginfast.c index 3f84e90b260..317041a454f 100644 --- a/src/backend/access/gin/ginfast.c +++ b/src/backend/access/gin/ginfast.c @@ -894,7 +894,7 @@ ginInsertCleanup(GinState *ginstate, bool full_clean, */ processPendingPage(&accum, &datums, page, FirstOffsetNumber); - vacuum_delay_point(); + vacuum_delay_point(false); /* * Is it time to flush memory to disk? Flush if we are at the end of @@ -931,7 +931,7 @@ ginInsertCleanup(GinState *ginstate, bool full_clean, { ginEntryInsert(ginstate, attnum, key, category, list, nlist, NULL); - vacuum_delay_point(); + vacuum_delay_point(false); } /* @@ -1004,7 +1004,7 @@ ginInsertCleanup(GinState *ginstate, bool full_clean, /* * Read next page in pending list */ - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBuffer(index, blkno); LockBuffer(buffer, GIN_SHARE); page = BufferGetPage(buffer); diff --git a/src/backend/access/gin/ginvacuum.c b/src/backend/access/gin/ginvacuum.c index a276eb020b5..9d0d5218392 100644 --- a/src/backend/access/gin/ginvacuum.c +++ b/src/backend/access/gin/ginvacuum.c @@ -663,12 +663,12 @@ ginbulkdelete(IndexVacuumInfo *info, IndexBulkDeleteResult *stats, UnlockReleaseBuffer(buffer); } - vacuum_delay_point(); + vacuum_delay_point(false); for (i = 0; i < nRoot; i++) { ginVacuumPostingTree(&gvs, rootOfPostingTree[i]); - vacuum_delay_point(); + vacuum_delay_point(false); } if (blkno == InvalidBlockNumber) /* rightmost page */ @@ -749,7 +749,7 @@ ginvacuumcleanup(IndexVacuumInfo *info, IndexBulkDeleteResult *stats) Buffer buffer; Page page; - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, info->strategy); diff --git a/src/backend/access/gist/gistvacuum.c b/src/backend/access/gist/gistvacuum.c index 0663193531a..8a9805207a8 100644 --- a/src/backend/access/gist/gistvacuum.c +++ b/src/backend/access/gist/gistvacuum.c @@ -277,7 +277,7 @@ gistvacuumpage(GistVacState *vstate, BlockNumber blkno, BlockNumber orig_blkno) recurse_to = InvalidBlockNumber; /* call vacuum_delay_point while not holding any buffer lock */ - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(rel, MAIN_FORKNUM, blkno, RBM_NORMAL, info->strategy); diff --git a/src/backend/access/hash/hash.c b/src/backend/access/hash/hash.c index 9ab8b829a1e..470daa022dc 100644 --- a/src/backend/access/hash/hash.c +++ b/src/backend/access/hash/hash.c @@ -730,7 +730,7 @@ hashbucketcleanup(Relation rel, Bucket cur_bucket, Buffer bucket_buf, bool retain_pin = false; bool clear_dead_marking = false; - vacuum_delay_point(); + vacuum_delay_point(false); page = BufferGetPage(buf); opaque = (HashPageOpaque) PageGetSpecialPointer(page); diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 44538b626c6..97c03036656 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -370,6 +370,10 @@ typedef struct LVRelState BlockNumber pages_removed; /* pages remove by truncation */ BlockNumber lpdead_item_pages; /* # pages with LP_DEAD items */ BlockNumber nonempty_pages; /* actually, last nonempty page + 1 */ + /* Counters reported as the relation's vacuum statistics */ + BlockNumber dead_pages; /* pages left with unremovable dead tuples */ + BlockNumber pages_frozen; /* pages where we froze tuples */ + BlockNumber pages_all_visible; /* pages we marked all-visible */ /* Statistics output by us, for table */ double new_rel_tuples; /* new estimated total # of tuples */ @@ -512,6 +516,10 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, write_rate; bool aggressive; /* should we scan all unfrozen pages? */ bool scanned_all_unfrozen; /* actually scanned all such pages? */ + bool freeze_age_vacuum; /* aggressive due to freeze age? */ + instr_time vacstart; + int64 startdelaytime; + instr_time vacend; char **indnames = NULL; TransactionId xidFullScanLimit; MultiXactId mxactFullScanLimit; @@ -527,11 +535,17 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, TransactionId FreezeLimit; MultiXactId MultiXactCutoff; + /* Used for instrumentation and cumulative maintenance statistics. */ + starttime = GetCurrentTimestamp(); + + /* measure elapsed and delay time for the vacuum statistics */ + INSTR_TIME_SET_CURRENT(vacstart); + startdelaytime = VacuumDelayTime; + /* measure elapsed time iff autovacuum logging requires it */ if (IsAutoVacuumWorkerProcess() && params->log_min_duration >= 0) { pg_rusage_init(&ru0); - starttime = GetCurrentTimestamp(); if (track_io_timing) { startreadtime = pgStatBlockReadTime; @@ -574,6 +588,14 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, xidFullScanLimit); aggressive |= MultiXactIdPrecedesOrEquals(rel->rd_rel->relminmxid, mxactFullScanLimit); + + /* + * Remember whether the freeze table age made this run aggressive. + * DISABLE_PAGE_SKIPPING can also force an aggressive scan, but does + * not by itself contribute to freeze_age_vacuum_count. + */ + freeze_age_vacuum = aggressive; + if (params->options & VACOPT_DISABLE_PAGE_SKIPPING) aggressive = true; @@ -752,7 +774,42 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, pgstat_report_vacuum(RelationGetRelid(rel), rel->rd_rel->relisshared, Max(new_live_tuples, 0), - vacrel->new_dead_tuples); + vacrel->new_dead_tuples, starttime, + VacuumDelayTime - startdelaytime, vacrel->failsafe_active); + + /* report the per-vacuum counters as well */ + { + PgStat_VacuumStats vacstats; + PgStat_Counter elapsedtime; + + MemSet(&vacstats, 0, sizeof(vacstats)); + vacstats.tuples_deleted = (PgStat_Counter) vacrel->tuples_deleted; + vacstats.dead_tuples = (PgStat_Counter) vacrel->new_dead_tuples; + vacstats.pages_deleted = (PgStat_Counter) vacrel->pages_removed; + vacstats.bytes_removed = (PgStat_Counter) vacrel->pages_removed * BLCKSZ; + vacstats.dead_pages = (PgStat_Counter) vacrel->dead_pages; + vacstats.pages_frozen = (PgStat_Counter) vacrel->pages_frozen; + vacstats.pages_all_visible = (PgStat_Counter) vacrel->pages_all_visible; + vacstats.freeze_age_vacuum_count = freeze_age_vacuum ? 1 : 0; + + INSTR_TIME_SET_CURRENT(vacend); + INSTR_TIME_SUBTRACT(vacend, vacstart); + elapsedtime = (PgStat_Counter) INSTR_TIME_GET_MICROSEC(vacend); + + ereport(elevel, + (errmsg("table \"%s\": vacuum statistics", vacrel->relname), + errdetail("elapsed: %.3f ms, cost-based delay: %.3f ms\n" + "aggressive scan required by freeze age: %s", + elapsedtime / 1000.0, + (VacuumDelayTime - startdelaytime) / 1000.0, + freeze_age_vacuum ? _("yes") : _("no")))); + + pgstat_report_vacstats(RelationGetRelid(rel), + rel->rd_rel->relisshared, + false, + &vacstats); + } + pgstat_progress_end_command(); /* and log the action if appropriate */ @@ -818,6 +875,14 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, vacrel->rel_pages, vacrel->pinskipped_pages, vacrel->frozenskipped_pages); + appendStringInfo(&buf, _("pages with dead tuples not yet removable: %u\n"), + vacrel->dead_pages); + appendStringInfo(&buf, _("pages with tuples frozen: %u\n"), + vacrel->pages_frozen); + appendStringInfo(&buf, _("pages marked all-visible: %u\n"), + vacrel->pages_all_visible); + appendStringInfo(&buf, _("aggressive scan required by freeze age: %s\n"), + freeze_age_vacuum ? _("yes") : _("no")); appendStringInfo(&buf, _("tuples: %lld removed, %lld remain, %lld are dead but not yet removable, oldest xmin: %u\n"), (long long) vacrel->tuples_deleted, @@ -858,8 +923,9 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, continue; appendStringInfo(&buf, - _("index \"%s\": pages: %u in total, %u newly deleted, %u currently deleted, %u reusable\n"), + _("index \"%s\": tuples: %.0f removed; pages: %u in total, %u newly deleted, %u currently deleted, %u reusable\n"), indnames[i], + istat->tuples_removed, istat->num_pages, istat->pages_newly_deleted, istat->pages_deleted, @@ -885,6 +951,8 @@ heap_vacuum_rel(Relation rel, VacuumParams *params, (long long) walusage.wal_records, (long long) walusage.wal_fpi, (unsigned long long) walusage.wal_bytes); + appendStringInfo(&buf, _("cost-based delay: %.3f ms\n"), + (VacuumDelayTime - startdelaytime) / 1000.0); appendStringInfo(&buf, _("system usage: %s"), pg_rusage_show(&ru0)); ereport(LOG, @@ -1075,7 +1143,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) if ((vmstatus & VISIBILITYMAP_ALL_VISIBLE) == 0) break; } - vacuum_delay_point(); + vacuum_delay_point(false); next_unskippable_block++; } } @@ -1127,7 +1195,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) if ((vmskipflags & VISIBILITYMAP_ALL_VISIBLE) == 0) break; } - vacuum_delay_point(); + vacuum_delay_point(false); next_unskippable_block++; } } @@ -1177,7 +1245,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) all_visible_according_to_vm = true; } - vacuum_delay_point(); + vacuum_delay_point(false); /* * Regularly check if wraparound failsafe should trigger. @@ -1391,6 +1459,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) visibilitymap_set(vacrel->rel, blkno, buf, InvalidXLogRecPtr, vmbuffer, InvalidTransactionId, VISIBILITYMAP_ALL_VISIBLE | VISIBILITYMAP_ALL_FROZEN); + vacrel->pages_all_visible++; END_CRIT_SECTION(); } @@ -1502,6 +1571,7 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) visibilitymap_set(vacrel->rel, blkno, buf, InvalidXLogRecPtr, vmbuffer, prunestate.visibility_cutoff_xid, flags); + vacrel->pages_all_visible++; } /* @@ -1685,6 +1755,12 @@ lazy_scan_heap(LVRelState *vacrel, VacuumParams *params, bool aggressive) appendStringInfo(&buf, _("%lld dead row versions cannot be removed yet, oldest xmin: %u\n"), (long long) vacrel->new_dead_tuples, vacrel->OldestXmin); + appendStringInfo(&buf, _("pages with dead tuples not yet removable: %u\n"), + vacrel->dead_pages); + appendStringInfo(&buf, _("pages with tuples frozen: %u\n"), + vacrel->pages_frozen); + appendStringInfo(&buf, _("pages marked all-visible: %u\n"), + vacrel->pages_all_visible); appendStringInfo(&buf, ngettext("Skipped %u page due to buffer pins, ", "Skipped %u pages due to buffer pins, ", vacrel->pinskipped_pages), @@ -1992,6 +2068,8 @@ lazy_scan_prune(LVRelState *vacrel, { Assert(prunestate->hastup); + vacrel->pages_frozen++; + /* * At least one tuple with storage needs to be frozen -- execute that * now. @@ -2092,6 +2170,10 @@ lazy_scan_prune(LVRelState *vacrel, dead_tuples->num_tuples); } + /* Remember pages that keep dead tuples we could not remove yet */ + if (new_dead_tuples > 0) + vacrel->dead_pages++; + /* Finally, add page-local counts to whole-VACUUM counts */ vacrel->tuples_deleted += tuples_deleted; vacrel->lpdead_items += lpdead_items; @@ -2374,7 +2456,7 @@ lazy_vacuum_heap_rel(LVRelState *vacrel) Page page; Size freespace; - vacuum_delay_point(); + vacuum_delay_point(false); tblk = ItemPointerGetBlockNumber(&vacrel->dead_tuples->itemptrs[tupindex]); vacrel->blkno = tblk; @@ -2544,8 +2626,12 @@ lazy_vacuum_heap_page(LVRelState *vacrel, BlockNumber blkno, Buffer buffer, Assert(BufferIsValid(*vmbuffer)); if (flags != 0) + { visibilitymap_set(vacrel->rel, blkno, buffer, InvalidXLogRecPtr, *vmbuffer, visibility_cutoff_xid, flags); + if (flags & VISIBILITYMAP_ALL_VISIBLE) + vacrel->pages_all_visible++; + } } /* Revert to the previous phase information for error traceback */ @@ -3044,6 +3130,81 @@ lazy_cleanup_all_indexes(LVRelState *vacrel) } } +/* + * lazy_index_vacstats_start() -- remember where an index vacuum call starts, + * for lazy_index_vacstats_finish(). + * + * The counters in istat accumulate over all the calls made for the index + * during one vacuum, so we report what each call added to them. That keeps + * the work of the bulk deletion passes accounted for even when the cleanup + * is skipped, and keeps it from being counted twice when it is not. + */ +static IndexBulkDeleteResult +lazy_index_vacstats_start(IndexBulkDeleteResult *istat, + instr_time *starttime, int64 *startdelaytime) +{ + IndexBulkDeleteResult before; + + if (istat) + before = *istat; + else + MemSet(&before, 0, sizeof(before)); + + INSTR_TIME_SET_CURRENT(*starttime); + *startdelaytime = VacuumDelayTime; + + return before; +} + +/* + * lazy_index_vacstats_finish() -- finish measuring one index vacuum call. + * + * pages_deleted and pages_free describe the whole index as the call left it, + * not what it did, so they are only looked at after the cleanup, which comes + * last: the deleted pages that are not reusable yet are the index's dead + * pages. + */ +static void +lazy_index_vacstats_finish(Relation indrel, IndexBulkDeleteResult *istat, + IndexBulkDeleteResult *before, bool cleanup, + instr_time starttime, int64 startdelaytime) +{ + PgStat_VacuumStats vacstats; + instr_time endtime; + + INSTR_TIME_SET_CURRENT(endtime); + INSTR_TIME_SUBTRACT(endtime, starttime); + + MemSet(&vacstats, 0, sizeof(vacstats)); + if (istat) + { + vacstats.tuples_deleted = + (PgStat_Counter) (istat->tuples_removed - before->tuples_removed); + /* + * Access methods accumulate pages_newly_deleted over the calls, + * so report only the work performed by this call. + */ + if (istat->pages_newly_deleted >= before->pages_newly_deleted) + vacstats.pages_deleted = (PgStat_Counter) + (istat->pages_newly_deleted - before->pages_newly_deleted); + else + vacstats.pages_deleted = (PgStat_Counter) istat->pages_newly_deleted; + if (cleanup && istat->pages_deleted > istat->pages_free) + vacstats.dead_pages = + (PgStat_Counter) (istat->pages_deleted - istat->pages_free); + } + + pgstat_report_index_vacuum_time(indrel, + (PgStat_Counter) INSTR_TIME_GET_MICROSEC(endtime), + VacuumDelayTime - startdelaytime, + IsAutoVacuumWorkerProcess()); + + pgstat_report_vacstats(RelationGetRelid(indrel), + indrel->rd_rel->relisshared, + true, + &vacstats); +} + /* * lazy_vacuum_one_index() -- vacuum index relation. * @@ -3062,6 +3223,9 @@ lazy_vacuum_one_index(Relation indrel, IndexBulkDeleteResult *istat, IndexVacuumInfo ivinfo; PGRUsage ru0; LVSavedErrInfo saved_err_info; + IndexBulkDeleteResult istat_before; + instr_time starttime; + int64 startdelaytime; pg_rusage_init(&ru0); @@ -3086,13 +3250,19 @@ lazy_vacuum_one_index(Relation indrel, IndexBulkDeleteResult *istat, InvalidBlockNumber, InvalidOffsetNumber); /* Do bulk deletion */ + istat_before = lazy_index_vacstats_start(istat, &starttime, + &startdelaytime); istat = index_bulk_delete(&ivinfo, istat, lazy_tid_reaped, (void *) vacrel->dead_tuples); + lazy_index_vacstats_finish(indrel, istat, &istat_before, false, + starttime, startdelaytime); ereport(elevel, (errmsg("scanned index \"%s\" to remove %d row versions", vacrel->indname, vacrel->dead_tuples->num_tuples), - errdetail_internal("%s", pg_rusage_show(&ru0)))); + errdetail("cost-based delay: %.3f ms\n%s", + (VacuumDelayTime - startdelaytime) / 1000.0, + pg_rusage_show(&ru0)))); /* Revert to the previous phase information for error traceback */ restore_vacuum_error_info(vacrel, &saved_err_info); @@ -3118,6 +3288,9 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat, IndexVacuumInfo ivinfo; PGRUsage ru0; LVSavedErrInfo saved_err_info; + IndexBulkDeleteResult istat_before; + instr_time starttime; + int64 startdelaytime; pg_rusage_init(&ru0); @@ -3142,7 +3315,11 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat, VACUUM_ERRCB_PHASE_INDEX_CLEANUP, InvalidBlockNumber, InvalidOffsetNumber); + istat_before = lazy_index_vacstats_start(istat, &starttime, + &startdelaytime); istat = index_vacuum_cleanup(&ivinfo, istat); + lazy_index_vacstats_finish(indrel, istat, &istat_before, true, + starttime, startdelaytime); if (istat) { @@ -3154,10 +3331,12 @@ lazy_cleanup_one_index(Relation indrel, IndexBulkDeleteResult *istat, errdetail("%.0f index row versions were removed.\n" "%u index pages were newly deleted.\n" "%u index pages are currently deleted, of which %u are currently reusable.\n" + "cost-based delay: %.3f ms\n" "%s.", (istat)->tuples_removed, (istat)->pages_newly_deleted, (istat)->pages_deleted, (istat)->pages_free, + (VacuumDelayTime - startdelaytime) / 1000.0, pg_rusage_show(&ru0)))); } @@ -4307,6 +4486,17 @@ vacuum_error_callback(void *arg) { LVRelState *errinfo = arg; + /* + * If an actual ERROR (not a lower-severity report that merely carries + * this vacuum error context) is being raised while we have a relation in + * hand, record at the database level that a vacuum was interrupted. Any + * error here aborts the vacuum, so the exact phase does not matter. We + * are inside the error handler, so this only bumps a counter; the + * statistics are updated at the next pgstat_report_stat(). + */ + if (errinfo->rel != NULL && geterrlevel() == ERROR) + pgstat_count_vacuum_error(errinfo->rel->rd_rel->relisshared); + switch (errinfo->phase) { case VACUUM_ERRCB_PHASE_SCAN_HEAP: diff --git a/src/backend/access/heap/visibilitymap.c b/src/backend/access/heap/visibilitymap.c index 7d252fa94ab..4663ade3aa3 100644 --- a/src/backend/access/heap/visibilitymap.c +++ b/src/backend/access/heap/visibilitymap.c @@ -90,6 +90,7 @@ #include "access/visibilitymap.h" #include "access/xlog.h" #include "miscadmin.h" +#include "pgstat.h" #include "port/pg_bitutils.h" #include "storage/bufmgr.h" #include "storage/lmgr.h" @@ -159,10 +160,24 @@ visibilitymap_clear(Relation rel, BlockNumber heapBlk, Buffer buf, uint8 flags) if (map[mapByte] & mask) { + uint8 cleared_bits = (map[mapByte] & mask) >> mapOffset; + map[mapByte] &= ~mask; MarkBufferDirty(buf); cleared = true; + + /* + * Count the pages that just lost their all-visible/all-frozen status + * for pg_stat_all_tables and pg_stat_database. + * The counters are delivered with the regular relation statistics, so + * nothing is counted during recovery, where rel is a fake relcache + * entry without a pgstat entry. + */ + if (cleared_bits & VISIBILITYMAP_ALL_VISIBLE) + pgstat_count_visible_page_marks_cleared(rel); + if (cleared_bits & VISIBILITYMAP_ALL_FROZEN) + pgstat_count_frozen_page_marks_cleared(rel); } LockBuffer(buf, BUFFER_LOCK_UNLOCK); diff --git a/src/backend/access/nbtree/nbtree.c b/src/backend/access/nbtree/nbtree.c index 8d4a587899c..d8e967402b2 100644 --- a/src/backend/access/nbtree/nbtree.c +++ b/src/backend/access/nbtree/nbtree.c @@ -1162,7 +1162,7 @@ btvacuumpage(BTVacState *vstate, BlockNumber scanblkno) backtrack_to = P_NONE; /* call vacuum_delay_point while not holding any buffer lock */ - vacuum_delay_point(); + vacuum_delay_point(false); /* * We can't use _bt_getbuf() here because it always applies diff --git a/src/backend/access/spgist/spgvacuum.c b/src/backend/access/spgist/spgvacuum.c index 76fb0374c42..8188dbedce0 100644 --- a/src/backend/access/spgist/spgvacuum.c +++ b/src/backend/access/spgist/spgvacuum.c @@ -613,14 +613,16 @@ spgvacuumpage(spgBulkDeleteState *bds, BlockNumber blkno) Relation index = bds->info->index; Buffer buffer; Page page; + bool was_empty; /* call vacuum_delay_point while not holding any buffer lock */ - vacuum_delay_point(); + vacuum_delay_point(false); buffer = ReadBufferExtended(index, MAIN_FORKNUM, blkno, RBM_NORMAL, bds->info->strategy); LockBuffer(buffer, BUFFER_LOCK_EXCLUSIVE); page = (Page) BufferGetPage(buffer); + was_empty = PageIsNew(page) || PageIsEmpty(page); if (PageIsNew(page)) { @@ -664,6 +666,8 @@ spgvacuumpage(spgBulkDeleteState *bds, BlockNumber blkno) { RecordFreeIndexPage(index, blkno); bds->stats->pages_deleted++; + if (!was_empty) + bds->stats->pages_newly_deleted++; } else { @@ -694,7 +698,7 @@ spgprocesspending(spgBulkDeleteState *bds) continue; /* ignore already-done items */ /* call vacuum_delay_point while not holding any buffer lock */ - vacuum_delay_point(); + vacuum_delay_point(false); /* examine the referenced page */ blkno = ItemPointerGetBlockNumber(&pitem->tid); @@ -891,7 +895,6 @@ spgvacuumscan(spgBulkDeleteState *bds) /* Report final stats */ bds->stats->num_pages = num_pages; - bds->stats->pages_newly_deleted = bds->stats->pages_deleted; bds->stats->pages_free = bds->stats->pages_deleted; } diff --git a/src/backend/commands/analyze.c b/src/backend/commands/analyze.c index 2630c4943f8..c1026843596 100644 --- a/src/backend/commands/analyze.c +++ b/src/backend/commands/analyze.c @@ -539,6 +539,9 @@ do_analyze_rel(Relation onerel, VacuumParams *params, save_sec_context | SECURITY_RESTRICTED_OPERATION); save_nestlevel = NewGUCNestLevel(); + /* Used for instrumentation and cumulative maintenance statistics. */ + starttime = GetCurrentTimestamp(); + /* measure elapsed time iff autovacuum logging requires it */ if (IsAutoVacuumWorkerProcess() && params->log_min_duration >= 0) { @@ -549,8 +552,6 @@ do_analyze_rel(Relation onerel, VacuumParams *params, } pg_rusage_init(&ru0); - if (params->log_min_duration >= 0) - starttime = GetCurrentTimestamp(); } /* @@ -1206,9 +1207,9 @@ do_analyze_rel(Relation onerel, VacuumParams *params, */ if (!inh) pgstat_report_analyze(onerel, totalrows, totaldeadrows, - (va_cols == NIL)); + (va_cols == NIL), starttime); else if (onerel->rd_rel->relkind == RELKIND_PARTITIONED_TABLE) - pgstat_report_analyze(onerel, 0, 0, (va_cols == NIL)); + pgstat_report_analyze(onerel, 0, 0, (va_cols == NIL), starttime); /* * If this isn't part of VACUUM ANALYZE, let index AMs do cleanup. @@ -1406,7 +1407,7 @@ compute_index_stats(Relation onerel, double totalrows, { HeapTuple heapTuple = rows[rowno]; - vacuum_delay_point(); + vacuum_delay_point(true); /* * Reset the per-tuple context each time, to reclaim any cruft @@ -1826,7 +1827,7 @@ acquire_sample_rows(Relation onerel, int elevel, prefetch_targblock = BlockSampler_Next(&prefetch_bs); #endif - vacuum_delay_point(); + vacuum_delay_point(true); block_accepted = table_scan_analyze_next_block(scan, targblock, vac_strategy); @@ -3452,7 +3453,7 @@ compute_trivial_stats(VacAttrStatsP stats, Datum value; bool isnull; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, i, &isnull); @@ -3574,7 +3575,7 @@ compute_distinct_stats(VacAttrStatsP stats, int firstcount1, j; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, i, &isnull); @@ -3934,7 +3935,7 @@ compute_scalar_stats(VacAttrStatsP stats, Datum value; bool isnull; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, i, &isnull); diff --git a/src/backend/commands/vacuum.c b/src/backend/commands/vacuum.c index 65cf10bb833..40b36c7b9bc 100644 --- a/src/backend/commands/vacuum.c +++ b/src/backend/commands/vacuum.c @@ -104,6 +104,15 @@ static MemoryContext vac_context = NULL; static BufferAccessStrategy vac_strategy; +/* + * Cumulative time this process has spent in cost-based VACUUM delays, in + * microseconds. Consumers take differences around an operation. ANALYZE + * uses the same delay function but does not contribute to this accumulator. + * Updated only while track_cost_delay_timing is enabled. + */ +int64 VacuumDelayTime = 0; +bool track_cost_delay_timing = false; + /* * Variables for cost-based parallel vacuum. See comments atop * compute_parallel_delay to understand how it works. @@ -2985,7 +2994,7 @@ vac_close_indexes(int nindexes, Relation *Irel, LOCKMODE lockmode) * typically once per page processed. */ void -vacuum_delay_point(void) +vacuum_delay_point(bool is_analyze) { double msec = 0; @@ -3007,13 +3016,28 @@ vacuum_delay_point(void) /* Nap if appropriate */ if (msec > 0) { + instr_time delay_start; + instr_time delay_end; + if (msec > VacuumCostDelay * 4) msec = VacuumCostDelay * 4; pgstat_report_wait_start(WAIT_EVENT_VACUUM_DELAY); + if (track_cost_delay_timing && !is_analyze) + INSTR_TIME_SET_CURRENT(delay_start); pg_usleep(msec * 1000); pgstat_report_wait_end(); + if (track_cost_delay_timing && !is_analyze) + { + int64 delay_us; + + INSTR_TIME_SET_CURRENT(delay_end); + INSTR_TIME_SUBTRACT(delay_end, delay_start); + delay_us = (int64) INSTR_TIME_GET_MICROSEC(delay_end); + VacuumDelayTime += delay_us; + } + /* * We don't want to ignore postmaster death during very long vacuums * with vacuum_cost_delay configured. We can't use the usual diff --git a/src/backend/commands/vacuum_ao.c b/src/backend/commands/vacuum_ao.c index dff6ecf332d..72435bd03ba 100644 --- a/src/backend/commands/vacuum_ao.c +++ b/src/backend/commands/vacuum_ao.c @@ -321,7 +321,18 @@ ao_vacuum_rel_post_cleanup(Relation onerel, VacuumParams *params, BufferAccessSt pgstat_report_vacuum(RelationGetRelid(onerel), onerel->rd_rel->relisshared, reltuples, - deadtuples); + deadtuples, + vacrelstats->starttime, + vacrelstats->delay_time + + (VacuumDelayTime - vacrelstats->phase_start_delay), + false); /* AO itself has no failsafe mode. */ + + /* + * Remember what is left behind for the vacuum statistics, which + * ao_vacuum_rel() reports once this last phase is over. + */ + vacrelstats->dead_tuples_left = (int64) deadtuples; + vacrelstats->total_file_segs = total_file_segs; SIMPLE_FAULT_INJECTOR("vacuum_ao_post_cleanup_end"); } @@ -429,11 +440,76 @@ init_vacrelstats() old_context = MemoryContextSwitchTo(TopMemoryContext); vacrelstats = (AOVacuumRelStats *) palloc0(sizeof(AOVacuumRelStats)); + /* Time the run from its first phase in this worker. */ + vacrelstats->starttime = GetCurrentTimestamp(); MemoryContextSwitchTo(old_context); return vacrelstats; } +/* + * Report what the vacuuming of an append-optimized relation did to the + * statistics collector, the way lazy vacuum does for a heap relation. + * + * The counters that describe heap pages have no counterpart here: an AO + * relation has no heap visibility map and nothing to freeze, so + * pages_frozen and pages_all_visible stay zero. Nor is there a relfrozenxid + * of its own to reach the freeze table age -- it is always invalid, the + * auxiliary heap relations being vacuumed and frozen on their own -- so the + * freeze_age_vacuum_count stays zero as well. What the heap reports as + * truncated pages is the space compaction freed, measured in blocks of the + * segment files it dropped or truncated. + * + * The indexes report themselves, from vacuum_appendonly_index() and scan_index(). + */ +static void +ao_report_vacuum_stats(Relation aorel, AOVacuumRelStats *vacrelstats) +{ + PgStat_VacuumStats vacstats; + + Assert(pgstat_track_vacuum_statistics); + + MemSet(&vacstats, 0, sizeof(vacstats)); + vacstats.tuples_deleted = (PgStat_Counter) vacrelstats->num_dead_tuples; + vacstats.dead_tuples = (PgStat_Counter) vacrelstats->dead_tuples_left; + /* Keep the conversion 64-bit: aggregate AO files can exceed BlockNumber. */ + vacstats.bytes_removed = vacrelstats->nbytes_truncated; + vacstats.pages_deleted = vacrelstats->nbytes_truncated / BLCKSZ + + (vacrelstats->nbytes_truncated % BLCKSZ != 0); + vacstats.total_file_segs = vacrelstats->total_file_segs; + + pgstat_report_vacstats(RelationGetRelid(aorel), + aorel->rd_rel->relisshared, + false, + &vacstats); +} + +/* Report index work, including cleanup runs with no obsolete AO segments. */ +static void +ao_report_index_vacuum_stats(Relation indexRelation, + IndexBulkDeleteResult *stats) +{ + PgStat_VacuumStats vacstats; + + Assert(pgstat_track_vacuum_statistics); + + MemSet(&vacstats, 0, sizeof(vacstats)); + if (stats) + { + vacstats.tuples_deleted = (PgStat_Counter) stats->tuples_removed; + vacstats.pages_deleted = (PgStat_Counter) stats->pages_newly_deleted; + /* deleted pages that are not yet reusable still hold dead entries */ + if (stats->pages_deleted > stats->pages_free) + vacstats.dead_pages = + (PgStat_Counter) (stats->pages_deleted - stats->pages_free); + } + + pgstat_report_vacstats(RelationGetRelid(indexRelation), + indexRelation->rd_rel->relisshared, + true, + &vacstats); +} + /* * ao_vacuum_rel() * @@ -443,6 +519,10 @@ void ao_vacuum_rel(Relation rel, VacuumParams *params, BufferAccessStrategy bstrategy) { static AOVacuumRelStats *vacrelstats = NULL; + instr_time phasestart; + instr_time phaseend; + int64 startdelaytime; + Assert(RelationStorageIsAO(rel)); Assert(params != NULL); @@ -469,20 +549,56 @@ ao_vacuum_rel(Relation rel, VacuumParams *params, BufferAccessStrategy bstrategy /* * Do the actual work --- either FULL or "lazy" vacuum + * + * Each phase is timed for the vacuum statistics, which are reported for + * the relation once the last phase is done. The phases run in separate + * transactions and may even end up in different vacuum workers; when that + * happens vacrelstats is reset above, and the statistics report describes + * the phases this worker did. */ + INSTR_TIME_SET_CURRENT(phasestart); + startdelaytime = VacuumDelayTime; + vacrelstats->phase_start_delay = startdelaytime; + if (ao_vacuum_phase == VACOPT_AO_PRE_CLEANUP_PHASE) ao_vacuum_rel_pre_cleanup(rel, params, bstrategy, vacrelstats); else if (ao_vacuum_phase == VACOPT_AO_COMPACT_PHASE) ao_vacuum_rel_compact(rel, params, bstrategy, vacrelstats); else if (ao_vacuum_phase == VACOPT_AO_POST_CLEANUP_PHASE) - { ao_vacuum_rel_post_cleanup(rel, params, bstrategy, vacrelstats); - pgstat_progress_end_command(); - cleanup_vacrelstats(&vacrelstats); - } else /* Do nothing here, we will launch the stages later */ Assert(ao_vacuum_phase == 0); + + INSTR_TIME_SET_CURRENT(phaseend); + INSTR_TIME_SUBTRACT(phaseend, phasestart); + vacrelstats->vacuum_time += (int64) INSTR_TIME_GET_MICROSEC(phaseend); + vacrelstats->delay_time += VacuumDelayTime - startdelaytime; + + if (ao_vacuum_phase == VACOPT_AO_POST_CLEANUP_PHASE) + { + int elevel = (params->options & VACOPT_VERBOSE) ? INFO : DEBUG2; + + if (Gp_role == GP_ROLE_DISPATCH) + elevel = DEBUG2; + + ereport(elevel, + (errmsg("append-optimized table \"%s\": vacuum statistics", + RelationGetRelationName(rel)), + errdetail("%lld dead tuples remain.\n" + "%lld bytes truncated; %lld file segments remain.\n" + "elapsed: %.3f ms, cost-based delay: %.3f ms", + (long long) vacrelstats->dead_tuples_left, + (long long) vacrelstats->nbytes_truncated, + (long long) vacrelstats->total_file_segs, + vacrelstats->vacuum_time / 1000.0, + vacrelstats->delay_time / 1000.0))); + + if (pgstat_track_vacuum_statistics) + ao_report_vacuum_stats(rel, vacrelstats); + pgstat_progress_end_command(); + cleanup_vacrelstats(&vacrelstats); + } } /* @@ -633,10 +749,15 @@ vacuum_appendonly_index(Relation indexRelation, IndexBulkDeleteResult *stats; IndexVacuumInfo ivinfo = {0}; PGRUsage ru0; + instr_time starttime; + instr_time endtime; + int64 startdelaytime; Assert(RelationIsValid(indexRelation)); pg_rusage_init(&ru0); + INSTR_TIME_SET_CURRENT(starttime); + startdelaytime = VacuumDelayTime; ivinfo.index = indexRelation; ivinfo.analyze_only = false; @@ -661,6 +782,17 @@ vacuum_appendonly_index(Relation indexRelation, /* Do post-VACUUM cleanup */ stats = index_vacuum_cleanup(&ivinfo, stats); + INSTR_TIME_SET_CURRENT(endtime); + INSTR_TIME_SUBTRACT(endtime, starttime); + + /* Ordinary timing follows track_counts, independently of work counters. */ + pgstat_report_index_vacuum_time(indexRelation, + (PgStat_Counter) INSTR_TIME_GET_MICROSEC(endtime), + VacuumDelayTime - startdelaytime, + IsAutoVacuumWorkerProcess()); + if (pgstat_track_vacuum_statistics) + ao_report_index_vacuum_stats(indexRelation, stats); + if (!stats) return; @@ -685,9 +817,11 @@ vacuum_appendonly_index(Relation indexRelation, stats->num_pages), errdetail("%.0f index row versions were removed.\n" "%u index pages have been deleted, %u are currently reusable.\n" + "cost-based delay: %.3f ms\n" "%s.", stats->tuples_removed, stats->pages_deleted, stats->pages_free, + (VacuumDelayTime - startdelaytime) / 1000.0, pg_rusage_show(&ru0)))); pfree(stats); @@ -806,8 +940,13 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac IndexBulkDeleteResult *stats; IndexVacuumInfo ivinfo = {0}; PGRUsage ru0; + instr_time starttime; + instr_time endtime; + int64 startdelaytime; pg_rusage_init(&ru0); + INSTR_TIME_SET_CURRENT(starttime); + startdelaytime = VacuumDelayTime; ivinfo.index = indrel; ivinfo.analyze_only = false; @@ -824,6 +963,17 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac /* Do post-VACUUM cleanup */ stats = index_vacuum_cleanup(&ivinfo, NULL); + INSTR_TIME_SET_CURRENT(endtime); + INSTR_TIME_SUBTRACT(endtime, starttime); + + /* Ordinary timing follows track_counts, independently of work counters. */ + pgstat_report_index_vacuum_time(indrel, + (PgStat_Counter) INSTR_TIME_GET_MICROSEC(endtime), + VacuumDelayTime - startdelaytime, + IsAutoVacuumWorkerProcess()); + if (pgstat_track_vacuum_statistics) + ao_report_index_vacuum_stats(indrel, stats); + if (!stats) return; @@ -847,8 +997,10 @@ scan_index(Relation indrel, Relation aorel, int elevel, BufferAccessStrategy vac stats->num_index_tuples, stats->num_pages), errdetail("%u index pages have been deleted, %u are currently reusable.\n" + "cost-based delay: %.3f ms\n" "%s.", stats->pages_deleted, stats->pages_free, + (VacuumDelayTime - startdelaytime) / 1000.0, pg_rusage_show(&ru0)))); pfree(stats); diff --git a/src/backend/postmaster/pgstat.c b/src/backend/postmaster/pgstat.c index 309a101fe02..b6f3592ff9a 100644 --- a/src/backend/postmaster/pgstat.c +++ b/src/backend/postmaster/pgstat.c @@ -39,6 +39,7 @@ #include "access/twophase_rmgr.h" #include "access/xact.h" #include "access/xlog.h" +#include "catalog/catalog.h" #include "catalog/pg_database.h" #include "catalog/pg_proc.h" #include "executor/instrument.h" @@ -126,6 +127,7 @@ * ---------- */ bool pgstat_track_counts = false; +bool pgstat_track_vacuum_statistics = false; int pgstat_track_functions = TRACK_FUNC_OFF; bool pgstat_collect_queuelevel = false; @@ -258,6 +260,10 @@ static PgStat_SubXactStatus *pgStatXactStack = NULL; static int pgStatXactCommit = 0; static int pgStatXactRollback = 0; + +/* Only incremented in error callbacks; sent by pgstat_report_stat(). */ +static PgStat_Counter pgStatVacuumErrors = 0; +static PgStat_Counter pgStatSharedVacuumErrors = 0; PgStat_Counter pgStatBlockReadTime = 0; PgStat_Counter pgStatBlockWriteTime = 0; static PgStat_Counter pgLastSessionReportTime = 0; @@ -372,6 +378,8 @@ static void pgstat_recv_resetslrucounter(PgStat_MsgResetslrucounter *msg, int le static void pgstat_recv_resetreplslotcounter(PgStat_MsgResetreplslotcounter *msg, int len); static void pgstat_recv_autovac(PgStat_MsgAutovacStart *msg, int len); static void pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len); +static void pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len); +static void pgstat_recv_resetvacstats(PgStat_MsgResetVacstats *msg, int len); static void pgstat_recv_analyze(PgStat_MsgAnalyze *msg, int len); static void pgstat_recv_archiver(PgStat_MsgArchiver *msg, int len); static void pgstat_recv_queuestat(PgStat_MsgQueuestat *msg, int len); /* GPDB */ @@ -893,6 +901,7 @@ pgstat_report_stat(bool disconnect) */ if ((pgStatTabList == NULL || pgStatTabList->tsa_used == 0) && pgStatXactCommit == 0 && pgStatXactRollback == 0 && + pgStatVacuumErrors == 0 && pgStatSharedVacuumErrors == 0 && pgWalUsage.wal_records == prevWalUsage.wal_records && WalStats.m_wal_write == 0 && WalStats.m_wal_sync == 0 && !have_function_stats && !disconnect) @@ -975,13 +984,14 @@ pgstat_report_stat(bool disconnect) /* * Send partial messages. Make sure that any pending xact commit/abort - * and connection stats get counted, even if there are no table stats to - * send. + * counts, connection stats and interrupted vacuums get counted, even if + * there are no table stats to send. */ if (regular_msg.m_nentries > 0 || - pgStatXactCommit > 0 || pgStatXactRollback > 0 || disconnect) + pgStatXactCommit > 0 || pgStatXactRollback > 0 || + pgStatVacuumErrors > 0 || disconnect) pgstat_send_tabstat(®ular_msg, now); - if (shared_msg.m_nentries > 0) + if (shared_msg.m_nentries > 0 || pgStatSharedVacuumErrors > 0) pgstat_send_tabstat(&shared_msg, now); /* Now, send function statistics */ @@ -1008,13 +1018,15 @@ pgstat_send_tabstat(PgStat_MsgTabstat *tsmsg, TimestampTz now) return; /* - * Report and reset accumulated xact commit/rollback and I/O timings - * whenever we send a normal tabstat message + * Report and reset accumulated xact commit/rollback, I/O timings and + * interrupted vacuums whenever we send a normal tabstat message. */ if (OidIsValid(tsmsg->m_databaseid)) { tsmsg->m_xact_commit = pgStatXactCommit; tsmsg->m_xact_rollback = pgStatXactRollback; + tsmsg->m_vacuum_interrupt_count = pgStatVacuumErrors; + pgStatVacuumErrors = 0; tsmsg->m_block_read_time = pgStatBlockReadTime; tsmsg->m_block_write_time = pgStatBlockWriteTime; @@ -1050,6 +1062,8 @@ pgstat_send_tabstat(PgStat_MsgTabstat *tsmsg, TimestampTz now) { tsmsg->m_xact_commit = 0; tsmsg->m_xact_rollback = 0; + tsmsg->m_vacuum_interrupt_count = pgStatSharedVacuumErrors; + pgStatSharedVacuumErrors = 0; tsmsg->m_block_read_time = 0; tsmsg->m_block_write_time = 0; tsmsg->m_session_time = 0; @@ -1589,7 +1603,9 @@ pgstat_report_autovac(Oid dboid) */ void pgstat_report_vacuum(Oid tableoid, bool shared, - PgStat_Counter livetuples, PgStat_Counter deadtuples) + PgStat_Counter livetuples, PgStat_Counter deadtuples, + TimestampTz starttime, PgStat_Counter delaytime, + bool failsafe) { PgStat_MsgVacuum msg; @@ -1600,12 +1616,112 @@ pgstat_report_vacuum(Oid tableoid, bool shared, msg.m_databaseid = shared ? InvalidOid : MyDatabaseId; msg.m_tableoid = tableoid; msg.m_autovacuum = IsAutoVacuumWorkerProcess(); + msg.m_isindex = false; + msg.m_failsafe = failsafe; + msg.m_delaytime = delaytime; msg.m_vacuumtime = GetCurrentTimestamp(); + msg.m_elapsedtime = Max(msg.m_vacuumtime - starttime, 0); msg.m_live_tuples = livetuples; msg.m_dead_tuples = deadtuples; pgstat_send(&msg, sizeof(msg)); } +/* + * Count a heap vacuum interrupted by ERROR. The caller is an error context + * callback, possibly running while a lock is held. Do not allocate memory, + * acquire locks or send messages here. Like transaction counts, these local + * counters are sent by pgstat_report_stat() outside a transaction. Shared + * relations belong to the InvalidOid database entry. + */ +void +pgstat_count_vacuum_error(bool shared) +{ + if (!pgstat_track_counts) + return; + + if (shared) + pgStatSharedVacuumErrors++; + else + pgStatVacuumErrors++; +} + +/* Report an index pass without changing table estimates or vacuum counts. */ +void +pgstat_report_index_vacuum_time(Relation rel, PgStat_Counter elapsedtime, + PgStat_Counter delaytime, bool is_autovacuum) +{ + PgStat_MsgVacuum msg; + + if (pgStatSock == PGINVALID_SOCKET || !pgstat_track_counts) + return; + + MemSet(&msg, 0, sizeof(msg)); + pgstat_setheader(&msg.m_hdr, PGSTAT_MTYPE_VACUUM); + msg.m_databaseid = rel->rd_rel->relisshared ? InvalidOid : MyDatabaseId; + msg.m_tableoid = RelationGetRelid(rel); + msg.m_autovacuum = is_autovacuum; + msg.m_isindex = true; + msg.m_elapsedtime = elapsedtime; + msg.m_delaytime = delaytime; + pgstat_send(&msg, sizeof(msg)); +} + +/* --------- + * pgstat_report_vacstats() - + * + * Tell the collector about the counters accumulated while vacuuming a + * relation (a table or an index). isindex tells which one, since only + * the tables count towards the per-database totals. + * --------- + */ +void +pgstat_report_vacstats(Oid tableoid, bool shared, bool isindex, + const PgStat_VacuumStats *stats) +{ + PgStat_MsgVacstats msg; + + if (pgStatSock == PGINVALID_SOCKET || !pgstat_track_counts || + !pgstat_track_vacuum_statistics) + return; + + pgstat_setheader(&msg.m_hdr, PGSTAT_MTYPE_VACSTATS); + msg.m_databaseid = shared ? InvalidOid : MyDatabaseId; + msg.m_tableoid = tableoid; + msg.m_isindex = isindex; + msg.m_stats = *stats; + pgstat_send(&msg, sizeof(msg)); +} + +/* ---------- + * pgstat_reset_vacuum_stats() - + * + * Tell the collector to throw away the vacuum counters of one relation of + * this database, or of all of them when resetall is true. + * ---------- + */ +void +pgstat_reset_vacuum_stats(Oid relid, bool resetall) +{ + PgStat_MsgResetVacstats msg; + + /* An invalid relation OID must never turn into a database-wide reset. */ + if (!resetall && !OidIsValid(relid)) + ereport(ERROR, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("invalid relation OID: %u", relid))); + Assert(!resetall || !OidIsValid(relid)); + + if (pgStatSock == PGINVALID_SOCKET) + return; + + pgstat_setheader(&msg.m_hdr, PGSTAT_MTYPE_RESETVACSTATS); + msg.m_databaseid = !resetall && IsSharedRelation(relid) ? + InvalidOid : MyDatabaseId; + msg.m_objectid = relid; + msg.m_resetall = resetall; + pgstat_send(&msg, sizeof(msg)); +} + /* -------- * pgstat_report_analyze() - * @@ -1618,7 +1734,7 @@ pgstat_report_vacuum(Oid tableoid, bool shared, void pgstat_report_analyze(Relation rel, PgStat_Counter livetuples, PgStat_Counter deadtuples, - bool resetcounter) + bool resetcounter, TimestampTz starttime) { PgStat_MsgAnalyze msg; @@ -1660,6 +1776,7 @@ pgstat_report_analyze(Relation rel, msg.m_autovacuum = IsAutoVacuumWorkerProcess(); msg.m_resetcounter = resetcounter; msg.m_analyzetime = GetCurrentTimestamp(); + msg.m_elapsedtime = Max(msg.m_analyzetime - starttime, 0); msg.m_live_tuples = livetuples; msg.m_dead_tuples = deadtuples; pgstat_send(&msg, sizeof(msg)); @@ -2805,6 +2922,25 @@ pgstat_fetch_stat_tabentry(Oid relid) } +/* ---------- + * pgstat_fetch_stat_vacuum_stats() - + * + * Return the vacuum counters available for a relation, or NULL. + * ---------- + */ +PgStat_VacuumStats * +pgstat_fetch_stat_vacuum_stats(Oid relid) +{ + PgStat_StatTabEntry *tabentry; + + if (!pgstat_track_vacuum_statistics) + return NULL; + + tabentry = pgstat_fetch_stat_tabentry(relid); + return tabentry ? &tabentry->vacuum_stats : NULL; +} + + /* ---------- * pgstat_fetch_stat_funcentry() - * @@ -3677,6 +3813,14 @@ PgstatCollectorMain(int argc, char *argv[]) pgstat_recv_vacuum(&msg.msg_vacuum, len); break; + case PGSTAT_MTYPE_VACSTATS: + pgstat_recv_vacstats(&msg.msg_vacstats, len); + break; + + case PGSTAT_MTYPE_RESETVACSTATS: + pgstat_recv_resetvacstats(&msg.msg_resetvacstats, len); + break; + case PGSTAT_MTYPE_ANALYZE: pgstat_recv_analyze(&msg.msg_analyze, len); break; @@ -3820,12 +3964,23 @@ reset_dbentry_counters(PgStat_StatDBEntry *dbentry) dbentry->n_sessions_abandoned = 0; dbentry->n_sessions_fatal = 0; dbentry->n_sessions_killed = 0; + dbentry->total_vacuum_time = 0; + dbentry->total_autovacuum_time = 0; + dbentry->total_vacuum_delay_time = 0; + dbentry->total_autovacuum_delay_time = 0; + dbentry->vacuum_failsafe_count = 0; + dbentry->vacuum_interrupt_count = 0; + + dbentry->n_frozen_page_marks_cleared = 0; + dbentry->n_visible_page_marks_cleared = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&dbentry->n_vacuum_stats, 0, sizeof(dbentry->n_vacuum_stats)); dbentry->stat_reset_timestamp = GetCurrentTimestamp(); dbentry->stats_timestamp = 0; hash_ctl.keysize = sizeof(Oid); - hash_ctl.entrysize = sizeof(PgStat_StatTabEntry); + hash_ctl.entrysize = PGSTAT_TAB_ENTRY_SIZE; dbentry->tables = hash_create("Per-database table", PGSTAT_TAB_HASH_SIZE, &hash_ctl, @@ -3914,6 +4069,17 @@ pgstat_get_tab_entry(PgStat_StatDBEntry *dbentry, Oid tableoid, bool create) result->analyze_count = 0; result->autovac_analyze_timestamp = 0; result->autovac_analyze_count = 0; + result->total_vacuum_time = 0; + result->total_autovacuum_time = 0; + result->total_analyze_time = 0; + result->total_autoanalyze_time = 0; + result->total_vacuum_delay_time = 0; + result->total_autovacuum_delay_time = 0; + result->vacuum_failsafe_count = 0; + result->frozen_page_marks_cleared = 0; + result->visible_page_marks_cleared = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&result->vacuum_stats, 0, sizeof(result->vacuum_stats)); } return result; @@ -4020,9 +4186,14 @@ pgstat_write_statsfiles(bool permanent, bool allDbs) * Write out the DB entry. We don't write the tables or functions * pointers, since they're of no use to any other process. */ - fputc('D', fpout); + fputc(pgstat_track_vacuum_statistics ? 'd' : 'D', fpout); rc = fwrite(dbentry, offsetof(PgStat_StatDBEntry, tables), 1, fpout); (void) rc; /* we'll check for error with ferror */ + if (pgstat_track_vacuum_statistics) + { + rc = fwrite(&dbentry->n_vacuum_stats, sizeof(PgStat_VacuumStats), 1, fpout); + (void) rc; + } } /* @@ -4169,8 +4340,8 @@ pgstat_write_db_statsfile(PgStat_StatDBEntry *dbentry, bool permanent) hash_seq_init(&tstat, dbentry->tables); while ((tabentry = (PgStat_StatTabEntry *) hash_seq_search(&tstat)) != NULL) { - fputc('T', fpout); - rc = fwrite(tabentry, sizeof(PgStat_StatTabEntry), 1, fpout); + fputc(pgstat_track_vacuum_statistics ? 't' : 'T', fpout); + rc = fwrite(tabentry, PGSTAT_TAB_ENTRY_SIZE, 1, fpout); (void) rc; /* we'll check for error with ferror */ } @@ -4256,6 +4427,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) HTAB *dbhash; FILE *fpin; int32 format_id; + int record_type; bool found; const char *statfile = permanent ? PGSTAT_STAT_PERMANENT_FILENAME : pgstat_stat_filename; int i; @@ -4273,7 +4445,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) * Create the DB hashtable */ hash_ctl.keysize = sizeof(Oid); - hash_ctl.entrysize = sizeof(PgStat_StatDBEntry); + hash_ctl.entrysize = PGSTAT_DB_ENTRY_SIZE; hash_ctl.hcxt = pgStatLocalContext; dbhash = hash_create("Databases hash", PGSTAT_DB_HASH_SIZE, &hash_ctl, HASH_ELEM | HASH_BLOBS | HASH_CONTEXT); @@ -4403,13 +4575,15 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) */ for (;;) { - switch (fgetc(fpin)) + switch (record_type = fgetc(fpin)) { /* - * 'D' A PgStat_StatDBEntry struct describing a database - * follows. + * 'D' Ordinary database counters follow. + * 'd' The same, followed by a PgStat_VacuumStats block. */ case 'D': + case 'd': + MemSet(&dbbuf, 0, sizeof(dbbuf)); if (fread(&dbbuf, 1, offsetof(PgStat_StatDBEntry, tables), fpin) != offsetof(PgStat_StatDBEntry, tables)) { @@ -4419,6 +4593,15 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) goto done; } + if (record_type == 'd' && + fread(&dbbuf.n_vacuum_stats, 1, sizeof(PgStat_VacuumStats), + fpin) != sizeof(PgStat_VacuumStats)) + { + ereport(pgStatRunningInCollector ? LOG : WARNING, + (errmsg("corrupted statistics file \"%s\"", statfile))); + goto done; + } + /* * Add to the DB hash */ @@ -4434,7 +4617,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) goto done; } - memcpy(dbentry, &dbbuf, sizeof(PgStat_StatDBEntry)); + memcpy(dbentry, &dbbuf, PGSTAT_DB_ENTRY_SIZE); dbentry->tables = NULL; dbentry->functions = NULL; @@ -4459,7 +4642,7 @@ pgstat_read_statsfiles(Oid onlydb, bool permanent, bool deep) } hash_ctl.keysize = sizeof(Oid); - hash_ctl.entrysize = sizeof(PgStat_StatTabEntry); + hash_ctl.entrysize = PGSTAT_TAB_ENTRY_SIZE; hash_ctl.hcxt = pgStatLocalContext; dbentry->tables = hash_create("Per-database table", PGSTAT_TAB_HASH_SIZE, @@ -4608,6 +4791,8 @@ pgstat_read_db_statsfile(Oid databaseid, HTAB *tabhash, HTAB *funchash, PgStat_StatFuncEntry *funcentry; FILE *fpin; int32 format_id; + int record_type; + size_t tabsize; bool found; char statfile[MAXPGPATH]; @@ -4649,14 +4834,18 @@ pgstat_read_db_statsfile(Oid databaseid, HTAB *tabhash, HTAB *funchash, */ for (;;) { - switch (fgetc(fpin)) + switch (record_type = fgetc(fpin)) { /* - * 'T' A PgStat_StatTabEntry follows. + * 'T' An ordinary table entry follows. + * 't' The entry also includes its vacuum counters. */ case 'T': - if (fread(&tabbuf, 1, sizeof(PgStat_StatTabEntry), - fpin) != sizeof(PgStat_StatTabEntry)) + case 't': + tabsize = record_type == 't' ? sizeof(PgStat_StatTabEntry) : + offsetof(PgStat_StatTabEntry, vacuum_stats); + MemSet(&tabbuf, 0, sizeof(tabbuf)); + if (fread(&tabbuf, 1, tabsize, fpin) != tabsize) { ereport(pgStatRunningInCollector ? LOG : WARNING, (errmsg("corrupted statistics file \"%s\"", @@ -4682,7 +4871,7 @@ pgstat_read_db_statsfile(Oid databaseid, HTAB *tabhash, HTAB *funchash, goto done; } - memcpy(tabentry, &tabbuf, sizeof(tabbuf)); + memcpy(tabentry, &tabbuf, PGSTAT_TAB_ENTRY_SIZE); break; /* @@ -4774,6 +4963,7 @@ pgstat_read_db_statsfile_timestamp(Oid databaseid, bool permanent, PgStat_StatReplSlotEntry myReplSlotStats; FILE *fpin; int32 format_id; + int record_type; const char *statfile = permanent ? PGSTAT_STAT_PERMANENT_FILENAME : pgstat_stat_filename; /* @@ -4857,13 +5047,15 @@ pgstat_read_db_statsfile_timestamp(Oid databaseid, bool permanent, */ for (;;) { - switch (fgetc(fpin)) + switch (record_type = fgetc(fpin)) { /* - * 'D' A PgStat_StatDBEntry struct describing a database - * follows. + * 'D' Ordinary database counters follow. + * 'd' The same, followed by a PgStat_VacuumStats block. */ case 'D': + case 'd': + MemSet(&dbentry, 0, sizeof(dbentry)); if (fread(&dbentry, 1, offsetof(PgStat_StatDBEntry, tables), fpin) != offsetof(PgStat_StatDBEntry, tables)) { @@ -4874,6 +5066,16 @@ pgstat_read_db_statsfile_timestamp(Oid databaseid, bool permanent, return false; } + if (record_type == 'd' && + fread(&dbentry.n_vacuum_stats, 1, sizeof(PgStat_VacuumStats), + fpin) != sizeof(PgStat_VacuumStats)) + { + ereport(pgStatRunningInCollector ? LOG : WARNING, + (errmsg("corrupted statistics file \"%s\"", statfile))); + FreeFile(fpin); + return false; + } + /* * If this is the DB we're looking for, save its timestamp and * we're done. @@ -5229,6 +5431,7 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) */ dbentry->n_xact_commit += (PgStat_Counter) (msg->m_xact_commit); dbentry->n_xact_rollback += (PgStat_Counter) (msg->m_xact_rollback); + dbentry->vacuum_interrupt_count += msg->m_vacuum_interrupt_count; dbentry->n_block_read_time += msg->m_block_read_time; dbentry->n_block_write_time += msg->m_block_write_time; @@ -5275,6 +5478,17 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) tabentry->analyze_count = 0; tabentry->autovac_analyze_timestamp = 0; tabentry->autovac_analyze_count = 0; + tabentry->total_vacuum_time = 0; + tabentry->total_autovacuum_time = 0; + tabentry->total_analyze_time = 0; + tabentry->total_autoanalyze_time = 0; + tabentry->total_vacuum_delay_time = 0; + tabentry->total_autovacuum_delay_time = 0; + tabentry->vacuum_failsafe_count = 0; + tabentry->frozen_page_marks_cleared = 0; + tabentry->visible_page_marks_cleared = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&tabentry->vacuum_stats, 0, sizeof(tabentry->vacuum_stats)); } else { @@ -5318,6 +5532,14 @@ pgstat_recv_tabstat(PgStat_MsgTabstat *msg, int len) dbentry->n_tuples_deleted += tabmsg->t_counts.t_tuples_deleted; dbentry->n_blocks_fetched += tabmsg->t_counts.t_blocks_fetched; dbentry->n_blocks_hit += tabmsg->t_counts.t_blocks_hit; + tabentry->frozen_page_marks_cleared += + tabmsg->t_counts.t_frozen_page_marks_cleared; + tabentry->visible_page_marks_cleared += + tabmsg->t_counts.t_visible_page_marks_cleared; + dbentry->n_frozen_page_marks_cleared += + tabmsg->t_counts.t_frozen_page_marks_cleared; + dbentry->n_visible_page_marks_cleared += + tabmsg->t_counts.t_visible_page_marks_cleared; } } @@ -5605,6 +5827,38 @@ pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len) tabentry = pgstat_get_tab_entry(dbentry, msg->m_tableoid, true); + if (msg->m_autovacuum) + { + tabentry->total_autovacuum_time += msg->m_elapsedtime; + tabentry->total_autovacuum_delay_time += msg->m_delaytime; + } + else + { + tabentry->total_vacuum_time += msg->m_elapsedtime; + tabentry->total_vacuum_delay_time += msg->m_delaytime; + } + + /* Index passes are already included in the owning table's elapsed time. */ + if (msg->m_isindex) + return; + + if (msg->m_failsafe) + { + tabentry->vacuum_failsafe_count++; + dbentry->vacuum_failsafe_count++; + } + + if (msg->m_autovacuum) + { + dbentry->total_autovacuum_time += msg->m_elapsedtime; + dbentry->total_autovacuum_delay_time += msg->m_delaytime; + } + else + { + dbentry->total_vacuum_time += msg->m_elapsedtime; + dbentry->total_vacuum_delay_time += msg->m_delaytime; + } + tabentry->n_live_tuples = msg->m_live_tuples; tabentry->n_dead_tuples = msg->m_dead_tuples; @@ -5632,6 +5886,120 @@ pgstat_recv_vacuum(PgStat_MsgVacuum *msg, int len) } } +/* ---------- + * pgstat_recv_vacstats() - + * + * Process a VACSTATS message: accumulate the vacuum counters into the + * relation's entry and, for a table, into the per-database totals. + * ---------- + */ +static void +pgstat_recv_vacstats(PgStat_MsgVacstats *msg, int len) +{ + PgStat_StatDBEntry *dbentry; + PgStat_VacuumStats *vacstats; + + if (!pgstat_track_vacuum_statistics) + return; + + dbentry = pgstat_get_db_entry(msg->m_databaseid, true); + vacstats = &pgstat_get_tab_entry(dbentry, msg->m_tableoid, true)->vacuum_stats; + + vacstats->tuples_deleted += msg->m_stats.tuples_deleted; + vacstats->dead_tuples += msg->m_stats.dead_tuples; + vacstats->pages_deleted += msg->m_stats.pages_deleted; + vacstats->bytes_removed += msg->m_stats.bytes_removed; + /* Relation state is replaced, never added to database totals. */ + vacstats->total_file_segs = msg->m_stats.total_file_segs; + vacstats->dead_pages += msg->m_stats.dead_pages; + vacstats->pages_frozen += msg->m_stats.pages_frozen; + vacstats->pages_all_visible += msg->m_stats.pages_all_visible; + vacstats->freeze_age_vacuum_count += + msg->m_stats.freeze_age_vacuum_count; + + /* + * The per-database totals describe what vacuum did to the tables. An + * index is vacuumed as a part of its table, and the time it took is + * already accounted for in the table's own report, so adding the index + * counters here would count that work twice. + */ + if (msg->m_isindex) + return; + + dbentry->n_vacuum_stats.tuples_deleted += msg->m_stats.tuples_deleted; + dbentry->n_vacuum_stats.dead_tuples += msg->m_stats.dead_tuples; + dbentry->n_vacuum_stats.pages_deleted += msg->m_stats.pages_deleted; + dbentry->n_vacuum_stats.bytes_removed += msg->m_stats.bytes_removed; + dbentry->n_vacuum_stats.dead_pages += msg->m_stats.dead_pages; + dbentry->n_vacuum_stats.pages_frozen += msg->m_stats.pages_frozen; + dbentry->n_vacuum_stats.pages_all_visible += msg->m_stats.pages_all_visible; + dbentry->n_vacuum_stats.freeze_age_vacuum_count += + msg->m_stats.freeze_age_vacuum_count; +} + +/* ---------- + * pgstat_recv_resetvacstats() - + * + * Throw away the vacuum counters of one relation, or of the whole database + * when no relation is given. This includes the VM revision counters; + * ordinary statistics are left alone. + * ---------- + */ +static void +pgstat_recv_resetvacstats(PgStat_MsgResetVacstats *msg, int len) +{ + PgStat_StatDBEntry *dbentry; + PgStat_StatTabEntry *tabentry; + HASH_SEQ_STATUS hstat; + + dbentry = pgstat_get_db_entry(msg->m_databaseid, false); + if (!dbentry) + return; + + if (!msg->m_resetall) + { + tabentry = pgstat_get_tab_entry(dbentry, msg->m_objectid, false); + if (tabentry != NULL) + { + tabentry->frozen_page_marks_cleared = 0; + tabentry->visible_page_marks_cleared = 0; + tabentry->total_vacuum_time = 0; + tabentry->total_autovacuum_time = 0; + tabentry->total_vacuum_delay_time = 0; + tabentry->total_autovacuum_delay_time = 0; + tabentry->vacuum_failsafe_count = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&tabentry->vacuum_stats, 0, sizeof(tabentry->vacuum_stats)); + } + return; + } + + hash_seq_init(&hstat, dbentry->tables); + while ((tabentry = (PgStat_StatTabEntry *) hash_seq_search(&hstat)) != NULL) + { + tabentry->frozen_page_marks_cleared = 0; + tabentry->visible_page_marks_cleared = 0; + tabentry->total_vacuum_time = 0; + tabentry->total_autovacuum_time = 0; + tabentry->total_vacuum_delay_time = 0; + tabentry->total_autovacuum_delay_time = 0; + tabentry->vacuum_failsafe_count = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&tabentry->vacuum_stats, 0, sizeof(tabentry->vacuum_stats)); + } + dbentry->n_frozen_page_marks_cleared = 0; + dbentry->n_visible_page_marks_cleared = 0; + dbentry->total_vacuum_time = 0; + dbentry->total_autovacuum_time = 0; + dbentry->total_vacuum_delay_time = 0; + dbentry->total_autovacuum_delay_time = 0; + dbentry->vacuum_failsafe_count = 0; + dbentry->vacuum_interrupt_count = 0; + if (pgstat_track_vacuum_statistics) + MemSet(&dbentry->n_vacuum_stats, 0, sizeof(dbentry->n_vacuum_stats)); +} + + /* ---------- * pgstat_recv_analyze() - * @@ -5666,11 +6034,13 @@ pgstat_recv_analyze(PgStat_MsgAnalyze *msg, int len) { tabentry->autovac_analyze_timestamp = msg->m_analyzetime; tabentry->autovac_analyze_count++; + tabentry->total_autoanalyze_time += msg->m_elapsedtime; } else { tabentry->analyze_timestamp = msg->m_analyzetime; tabentry->analyze_count++; + tabentry->total_analyze_time += msg->m_elapsedtime; } } diff --git a/src/backend/tsearch/ts_typanalyze.c b/src/backend/tsearch/ts_typanalyze.c index 504ba1569ee..5c5ff6d8dc1 100644 --- a/src/backend/tsearch/ts_typanalyze.c +++ b/src/backend/tsearch/ts_typanalyze.c @@ -206,7 +206,7 @@ compute_tsvector_stats(VacAttrStats *stats, char *lexemesptr; int j; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, vector_no, &isnull); diff --git a/src/backend/utils/adt/array_typanalyze.c b/src/backend/utils/adt/array_typanalyze.c index 8993d23e18b..71f99bbd327 100644 --- a/src/backend/utils/adt/array_typanalyze.c +++ b/src/backend/utils/adt/array_typanalyze.c @@ -314,7 +314,7 @@ compute_array_stats(VacAttrStats *stats, AnalyzeAttrFetchFunc fetchfunc, int distinct_count; bool count_item_found; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, array_no, &isnull); if (isnull) diff --git a/src/backend/utils/adt/rangetypes_typanalyze.c b/src/backend/utils/adt/rangetypes_typanalyze.c index 9d5cf897c45..444c6c2da8e 100644 --- a/src/backend/utils/adt/rangetypes_typanalyze.c +++ b/src/backend/utils/adt/rangetypes_typanalyze.c @@ -168,7 +168,7 @@ compute_range_stats(VacAttrStats *stats, AnalyzeAttrFetchFunc fetchfunc, upper; float8 length; - vacuum_delay_point(); + vacuum_delay_point(true); value = fetchfunc(stats, range_no, &isnull); if (isnull) diff --git a/src/backend/utils/error/elog.c b/src/backend/utils/error/elog.c index 6c8db2ef5fa..e5566c413ae 100644 --- a/src/backend/utils/error/elog.c +++ b/src/backend/utils/error/elog.c @@ -1647,6 +1647,23 @@ geterrcode(void) return edata->sqlerrcode; } +/* + * geterrlevel --- return the elevel of the error currently being constructed + * + * This is only intended for use in error callback subroutines, where it lets + * a callback tell a genuine error apart from a lower-severity report. + */ +int +geterrlevel(void) +{ + ErrorData *edata = &errordata[errordata_stack_depth]; + + /* we don't bother incrementing recursion_depth */ + CHECK_STACK_DEPTH(); + + return edata->elevel; +} + /* * geterrposition --- return the currently set error position (0 if none) * diff --git a/src/backend/utils/misc/guc.c b/src/backend/utils/misc/guc.c index 6166c5ff249..ef548d23930 100644 --- a/src/backend/utils/misc/guc.c +++ b/src/backend/utils/misc/guc.c @@ -1627,6 +1627,24 @@ static struct config_bool ConfigureNamesBool[] = true, NULL, NULL, NULL }, + { + {"track_vacuum_statistics", PGC_POSTMASTER, STATS_COLLECTOR, + gettext_noop("Collects statistics on what vacuum did and what it cost."), + gettext_noop("The counters are exposed by the vacuum_stats extension.") + }, + &pgstat_track_vacuum_statistics, + false, + NULL, NULL, NULL + }, + { + {"track_cost_delay_timing", PGC_SUSET, STATS_COLLECTOR, + gettext_noop("Collects timing statistics for cost-based vacuum delay."), + NULL + }, + &track_cost_delay_timing, + false, + NULL, NULL, NULL + }, { {"track_io_timing", PGC_SUSET, STATS_COLLECTOR, gettext_noop("Collects timing statistics for database I/O activity."), diff --git a/src/backend/utils/misc/postgresql.conf.sample b/src/backend/utils/misc/postgresql.conf.sample index 4192dfb2748..0487835dd38 100644 --- a/src/backend/utils/misc/postgresql.conf.sample +++ b/src/backend/utils/misc/postgresql.conf.sample @@ -626,6 +626,8 @@ optimizer_analyze_root_partition = on # stats collection on root partitions #track_activities = on #track_activity_query_size = 1024 # (change requires restart) #track_counts = off +#track_vacuum_statistics = off # collect vacuum work and cost (change requires restart) +#track_cost_delay_timing = off #track_io_timing = off #track_wal_io_timing = off #track_functions = none # none, pl, all diff --git a/src/include/access/appendonly_compaction.h b/src/include/access/appendonly_compaction.h index 44aa78a39fb..59a552adfee 100644 --- a/src/include/access/appendonly_compaction.h +++ b/src/include/access/appendonly_compaction.h @@ -13,6 +13,7 @@ #ifndef APPENDONLY_COMPACTION_H #define APPENDONLY_COMPACTION_H +#include "datatype/timestamp.h" #include "nodes/pg_list.h" #include "access/appendonly_visimap.h" #include "utils/rel.h" @@ -28,9 +29,17 @@ */ typedef struct AOVacuumRelStats { - int nbytes_truncated; /* current # of bytes truncated from segment file */ - int num_dead_tuples; /* current # of dead tuples */ + TimestampTz starttime; /* start of the first vacuum phase in this worker */ + int64 nbytes_truncated; /* current # of bytes truncated from segment file */ + int64 num_dead_tuples; /* current # of dead tuples */ int num_index_vacuumed; /* current # of indexes been vacuumed */ + + /* for the vacuum statistics, accumulated over all the phases */ + int64 vacuum_time; /* time spent in the phases, in microseconds */ + int64 phase_start_delay; /* counter at the start of the current phase */ + int64 delay_time; /* of which the cost-based vacuum delay */ + int64 dead_tuples_left; /* tuples the post-cleanup found still hidden */ + int64 total_file_segs; /* segment metadata entries after post-cleanup */ } AOVacuumRelStats; extern Bitmapset *AppendOptimizedCollectDeadSegments(Relation aorel); diff --git a/src/include/commands/vacuum.h b/src/include/commands/vacuum.h index e33d39973c0..7b80970f0cd 100644 --- a/src/include/commands/vacuum.h +++ b/src/include/commands/vacuum.h @@ -371,6 +371,8 @@ extern int vacuum_multixact_failsafe_age; extern pg_atomic_uint32 *VacuumSharedCostBalance; extern pg_atomic_uint32 *VacuumActiveNWorkers; extern int VacuumCostBalanceLocal; +extern PGDLLIMPORT int64 VacuumDelayTime; +extern PGDLLIMPORT bool track_cost_delay_timing; /* in commands/vacuum.c */ @@ -409,7 +411,7 @@ extern void vacuum_set_xid_limits(Relation rel, extern bool vacuum_xid_failsafe_check(TransactionId relfrozenxid, MultiXactId relminmxid); extern void vac_update_datfrozenxid(void); -extern void vacuum_delay_point(void); +extern void vacuum_delay_point(bool is_analyze); extern bool vacuum_is_relation_owner(Oid relid, Form_pg_class reltuple, bits32 options); extern Relation vacuum_open_relation(Oid relid, RangeVar *relation, diff --git a/src/include/pgstat.h b/src/include/pgstat.h index c54b1bb369a..717090a07dc 100644 --- a/src/include/pgstat.h +++ b/src/include/pgstat.h @@ -85,6 +85,8 @@ typedef enum StatMsgType PGSTAT_MTYPE_REPLSLOT, PGSTAT_MTYPE_CONNECT, PGSTAT_MTYPE_DISCONNECT, + PGSTAT_MTYPE_VACSTATS, + PGSTAT_MTYPE_RESETVACSTATS, } StatMsgType; /* ---------- @@ -133,6 +135,9 @@ typedef struct PgStat_TableCounts PgStat_Counter t_blocks_fetched; PgStat_Counter t_blocks_hit; + + PgStat_Counter t_frozen_page_marks_cleared; + PgStat_Counter t_visible_page_marks_cleared; } PgStat_TableCounts; /* Possible targets for resetting cluster-wide shared values */ @@ -282,7 +287,7 @@ typedef struct PgStat_TableEntry * ---------- */ #define PGSTAT_NUM_TABENTRIES \ - ((PGSTAT_MSG_PAYLOAD - sizeof(Oid) - 3 * sizeof(int) - 5 * sizeof(PgStat_Counter)) \ + ((PGSTAT_MSG_PAYLOAD - sizeof(Oid) - 3 * sizeof(int) - 6 * sizeof(PgStat_Counter)) \ / sizeof(PgStat_TableEntry)) typedef struct PgStat_MsgTabstat @@ -297,6 +302,7 @@ typedef struct PgStat_MsgTabstat PgStat_Counter m_session_time; PgStat_Counter m_active_time; PgStat_Counter m_idle_in_xact_time; + PgStat_Counter m_vacuum_interrupt_count; PgStat_TableEntry m_entry[PGSTAT_NUM_TABENTRIES]; } PgStat_MsgTabstat; @@ -413,12 +419,55 @@ typedef struct PgStat_MsgVacuum Oid m_databaseid; Oid m_tableoid; bool m_autovacuum; + bool m_isindex; /* index time does not update tuple/count fields or DB totals */ TimestampTz m_vacuumtime; PgStat_Counter m_live_tuples; PgStat_Counter m_dead_tuples; + PgStat_Counter m_elapsedtime; /* microseconds */ + PgStat_Counter m_delaytime; /* microseconds */ + bool m_failsafe; /* this completed table vacuum entered failsafe */ } PgStat_MsgVacuum; +/* ---------- + * PgStat_VacuumStats Vacuum statistics reported for a relation. + * ---------- + */ +typedef struct PgStat_VacuumStats +{ + PgStat_Counter tuples_deleted; /* tuples removed by vacuum */ + PgStat_Counter dead_tuples; /* dead tuples left unremoved */ + PgStat_Counter pages_deleted; /* pages removed/deleted by vacuum */ + PgStat_Counter bytes_removed; /* bytes physically truncated from table files */ + PgStat_Counter dead_pages; /* pages with unremoved dead tuples */ + PgStat_Counter total_file_segs; /* latest AO segment count, not cumulative */ + PgStat_Counter pages_frozen; /* pages where vacuum froze tuples */ + PgStat_Counter pages_all_visible; /* pages marked all-visible by vacuum */ + + /* + * Number of heap vacuum runs made aggressive by the XID or MultiXact + * freeze table age. This includes VACUUM FREEZE, which sets those + * age thresholds to zero, but not DISABLE_PAGE_SKIPPING alone. + */ + PgStat_Counter freeze_age_vacuum_count; +} PgStat_VacuumStats; + +/* ---------- + * PgStat_MsgVacstats Sent by the backend or autovacuum daemon + * after vacuuming a heap relation or an index + * to report per-relation vacuum counters. + * ---------- + */ +typedef struct PgStat_MsgVacstats +{ + PgStat_MsgHdr m_hdr; + Oid m_databaseid; + Oid m_tableoid; + bool m_isindex; /* counted apart from the database totals */ + PgStat_VacuumStats m_stats; +} PgStat_MsgVacstats; + + /* ---------- * PgStat_MsgAnalyze Sent by the backend or autovacuum daemon * after ANALYZE @@ -434,6 +483,7 @@ typedef struct PgStat_MsgAnalyze TimestampTz m_analyzetime; PgStat_Counter m_live_tuples; PgStat_Counter m_dead_tuples; + PgStat_Counter m_elapsedtime; /* microseconds */ } PgStat_MsgAnalyze; @@ -685,6 +735,20 @@ typedef struct PgStat_MsgDisconnect SessionEndType m_cause; } PgStat_MsgDisconnect; +/* ---------- + * PgStat_MsgResetVacstats Sent by the backend to throw away the vacuum + * counters of one relation, or of the whole + * database when m_resetall is true. + * ---------- + */ +typedef struct PgStat_MsgResetVacstats +{ + PgStat_MsgHdr m_hdr; + Oid m_databaseid; + Oid m_objectid; + bool m_resetall; +} PgStat_MsgResetVacstats; + /* ---------- * PgStat_Msg Union over all possible messages. * ---------- @@ -704,6 +768,8 @@ typedef union PgStat_Msg PgStat_MsgResetreplslotcounter msg_resetreplslotcounter; PgStat_MsgAutovacStart msg_autovacuum_start; PgStat_MsgVacuum msg_vacuum; + PgStat_MsgVacstats msg_vacstats; + PgStat_MsgResetVacstats msg_resetvacstats; PgStat_MsgAnalyze msg_analyze; PgStat_MsgArchiver msg_archiver; PgStat_MsgQueuestat msg_queuestat; /* GPDB */ @@ -730,7 +796,7 @@ typedef union PgStat_Msg * ------------------------------------------------------------ */ -#define PGSTAT_FILE_FORMAT_ID 0x01A5BCA2 +#define PGSTAT_FILE_FORMAT_ID 0x01A5BCAD /* ---------- * PgStat_StatDBEntry The collector's data per database @@ -769,15 +835,32 @@ typedef struct PgStat_StatDBEntry PgStat_Counter n_sessions_fatal; PgStat_Counter n_sessions_killed; + /* Cumulative table vacuum times; index work is already included. */ + PgStat_Counter total_vacuum_time; /* microseconds */ + PgStat_Counter total_autovacuum_time; /* microseconds */ + PgStat_Counter total_vacuum_delay_time; /* microseconds */ + PgStat_Counter total_autovacuum_delay_time; /* microseconds */ + PgStat_Counter vacuum_failsafe_count; + + /* Heap vacuums in this database interrupted by ERROR. */ + PgStat_Counter vacuum_interrupt_count; + + /* VM revisions are fed by ordinary relation statistics. */ + PgStat_Counter n_frozen_page_marks_cleared; + PgStat_Counter n_visible_page_marks_cleared; + TimestampTz stat_reset_timestamp; TimestampTz stats_timestamp; /* time of db stats file update */ /* - * tables and functions must be last in the struct, because we don't write - * the pointers out to the stats file. + * Only the prefix before these pointers is written as ordinary database + * statistics. The optional vacuum counters are serialized separately. */ HTAB *tables; HTAB *functions; + + /* Must be last: storage is omitted when tracking is disabled at startup. */ + PgStat_VacuumStats n_vacuum_stats; } PgStat_StatDBEntry; @@ -816,8 +899,32 @@ typedef struct PgStat_StatTabEntry PgStat_Counter analyze_count; TimestampTz autovac_analyze_timestamp; /* autovacuum initiated */ PgStat_Counter autovac_analyze_count; + + /* Cumulative maintenance times, in microseconds. */ + PgStat_Counter total_vacuum_time; + PgStat_Counter total_autovacuum_time; + PgStat_Counter total_analyze_time; + PgStat_Counter total_autoanalyze_time; + PgStat_Counter total_vacuum_delay_time; + PgStat_Counter total_autovacuum_delay_time; + PgStat_Counter vacuum_failsafe_count; + + /* VM revisions are fed by ordinary relation statistics. */ + PgStat_Counter frozen_page_marks_cleared; + PgStat_Counter visible_page_marks_cleared; + + /* Must be last: storage is omitted when tracking is disabled at startup. */ + PgStat_VacuumStats vacuum_stats; } PgStat_StatTabEntry; +/* The postmaster setting fixes hash entry sizes for the process lifetime. */ +#define PGSTAT_DB_ENTRY_SIZE \ + (pgstat_track_vacuum_statistics ? sizeof(PgStat_StatDBEntry) : \ + offsetof(PgStat_StatDBEntry, n_vacuum_stats)) +#define PGSTAT_TAB_ENTRY_SIZE \ + (pgstat_track_vacuum_statistics ? sizeof(PgStat_StatTabEntry) : \ + offsetof(PgStat_StatTabEntry, vacuum_stats)) + /* ---------- * PgStat_StatQueueEntry The collector's data per resource queue @@ -986,6 +1093,7 @@ typedef struct PgStat_FunctionCallUsage * ---------- */ extern PGDLLIMPORT bool pgstat_track_counts; +extern PGDLLIMPORT bool pgstat_track_vacuum_statistics; extern PGDLLIMPORT int pgstat_track_functions; extern char *pgstat_stat_directory; extern char *pgstat_stat_tmpname; @@ -1059,10 +1167,32 @@ extern void pgstat_reset_replslot_counter(const char *name); extern void pgstat_report_connect(Oid dboid); extern void pgstat_report_autovac(Oid dboid); extern void pgstat_report_vacuum(Oid tableoid, bool shared, - PgStat_Counter livetuples, PgStat_Counter deadtuples); + PgStat_Counter livetuples, PgStat_Counter deadtuples, + TimestampTz starttime, PgStat_Counter delaytime, + bool failsafe); +extern void pgstat_count_vacuum_error(bool shared); +extern void pgstat_report_index_vacuum_time(Relation rel, + PgStat_Counter elapsedtime, + PgStat_Counter delaytime, bool is_autovacuum); +extern void pgstat_report_vacstats(Oid tableoid, bool shared, bool isindex, + const PgStat_VacuumStats *stats); +extern void pgstat_reset_vacuum_stats(Oid relid, bool resetall); + +/* count a page whose all-visible bit is being cleared */ +#define pgstat_count_visible_page_marks_cleared(rel) \ + do { \ + if ((rel)->pgstat_info != NULL) \ + (rel)->pgstat_info->t_counts.t_visible_page_marks_cleared++; \ + } while (0) +/* count a page whose all-frozen bit is being cleared */ +#define pgstat_count_frozen_page_marks_cleared(rel) \ + do { \ + if ((rel)->pgstat_info != NULL) \ + (rel)->pgstat_info->t_counts.t_frozen_page_marks_cleared++; \ + } while (0) extern void pgstat_report_analyze(Relation rel, PgStat_Counter livetuples, PgStat_Counter deadtuples, - bool resetcounter); + bool resetcounter, TimestampTz starttime); extern void pgstat_report_recovery_conflict(int reason); extern void pgstat_report_deadlock(void); @@ -1270,6 +1400,7 @@ extern void pgstat_combine_from_qe(struct CdbDispatchResults *results, /* GPDB * */ extern PgStat_StatDBEntry *pgstat_fetch_stat_dbentry(Oid dbid); extern PgStat_StatTabEntry *pgstat_fetch_stat_tabentry(Oid relid); +extern PgStat_VacuumStats *pgstat_fetch_stat_vacuum_stats(Oid relid); extern PgStat_StatQueueEntry *pgstat_fetch_stat_queueentry(Oid queueid); /* GPDB */ extern PgBackendStatus *pgstat_fetch_stat_beentry(int beid); diff --git a/src/include/utils/elog.h b/src/include/utils/elog.h index 001bc08a3a5..97f218e907c 100644 --- a/src/include/utils/elog.h +++ b/src/include/utils/elog.h @@ -263,6 +263,7 @@ extern void internalerrquery(const char *query); extern void err_generic_string(int field, const char *str); extern int geterrcode(void); +extern int geterrlevel(void); extern int geterrposition(void); extern int getinternalerrposition(void); diff --git a/src/include/utils/unsync_guc_name.h b/src/include/utils/unsync_guc_name.h index 85ecb3548e6..9b98358a478 100644 --- a/src/include/utils/unsync_guc_name.h +++ b/src/include/utils/unsync_guc_name.h @@ -599,9 +599,11 @@ "track_activities", "track_activity_query_size", "track_commit_timestamp", + "track_cost_delay_timing", "track_counts", "track_functions", "track_io_timing", + "track_vacuum_statistics", "transaction_deferrable", "transaction_isolation", "transaction_read_only",