From 4590b63c7c88ed2372dc5be2be02227e47841161 Mon Sep 17 00:00:00 2001 From: Bharath Rupireddy Date: Sun, 27 Sep 2026 00:07:24 +0000 Subject: [PATCH v1] Fix leader's error context during parallel index vacuuming. Commit 8e1fae1938 moved the parallel vacuum code to vacuumparallel.c, and since then the leader processes indexes through parallel_vacuum_process_one_index() just like the workers do. Only the workers, however, install the error context callback that reports the index being worked on. The leader keeps the lazy vacuum error callback with the phase left over from the heap scan, so an error it raises while vacuuming or cleaning up an index reports the heap instead of the index, as "while scanning block N of relation" for an index pass in the middle of the scan or as "while scanning relation" for the pass after it. Errors relayed from the workers get the same stale line appended. Before 8e1fae1938 the leader went through lazy_vacuum_one_index() and lazy_cleanup_one_index(), which set the phase and the index name. The log is often the only evidence there is when an index page turns out to be bad, and with parallel autovacuum in PG19 every autovacuum of a large table goes through this path. Fix this by installing the parallel vacuum error callback in the leader for as long as it processes indexes itself, and by filling in the relation names it needs when the parallel state is created. The callers set the lazy vacuum phase aside for the duration of the parallel index phases so that the heap scan line is not reported alongside it. This is an inconsistency in the context that gets reported rather than a bug, so it is not back-patched. Oversight in commit 8e1fae1938. Reported-by: Claude Code Author: Bharath Rupireddy Discussion: https://postgr.es/m/<> --- src/backend/access/heap/vacuumlazy.c | 23 ++++++- src/backend/commands/vacuumparallel.c | 26 ++++++- src/test/modules/nbtree/Makefile | 3 +- .../expected/nbtree_vacuum_error_context.out | 68 +++++++++++++++++++ src/test/modules/nbtree/meson.build | 1 + .../sql/nbtree_vacuum_error_context.sql | 59 ++++++++++++++++ 6 files changed, 175 insertions(+), 5 deletions(-) create mode 100644 src/test/modules/nbtree/expected/nbtree_vacuum_error_context.out create mode 100644 src/test/modules/nbtree/sql/nbtree_vacuum_error_context.sql diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c index 997d84a77b3..f5a11bac026 100644 --- a/src/backend/access/heap/vacuumlazy.c +++ b/src/backend/access/heap/vacuumlazy.c @@ -2564,10 +2564,20 @@ lazy_vacuum_all_indexes(LVRelState *vacrel) } else { - /* Outsource everything to parallel variant */ + LVSavedErrInfo saved_err_info; + + /* + * Outsource everything to parallel variant. An error raised while the + * leader itself processes an index gets its context from the parallel + * vacuum code, so stop reporting the heap phase we came from. + */ + update_vacuum_error_info(vacrel, &saved_err_info, + VACUUM_ERRCB_PHASE_UNKNOWN, + InvalidBlockNumber, InvalidOffsetNumber); parallel_vacuum_bulkdel_all_indexes(vacrel->pvs, old_live_tuples, vacrel->num_index_scans, &(vacrel->worker_usage.vacuum)); + restore_vacuum_error_info(vacrel, &saved_err_info); /* * Do a postcheck to consider applying wraparound failsafe now. Note @@ -3008,11 +3018,20 @@ lazy_cleanup_all_indexes(LVRelState *vacrel) } else { - /* Outsource everything to parallel variant */ + LVSavedErrInfo saved_err_info; + + /* + * Outsource everything to parallel variant. See + * lazy_vacuum_all_indexes() for the error context handling. + */ + update_vacuum_error_info(vacrel, &saved_err_info, + VACUUM_ERRCB_PHASE_UNKNOWN, + InvalidBlockNumber, InvalidOffsetNumber); parallel_vacuum_cleanup_all_indexes(vacrel->pvs, reltuples, vacrel->num_index_scans, estimated_count, &(vacrel->worker_usage.cleanup)); + restore_vacuum_error_info(vacrel, &saved_err_info); } /* Reset the progress counters */ diff --git a/src/backend/commands/vacuumparallel.c b/src/backend/commands/vacuumparallel.c index 4532da60c84..f657494a496 100644 --- a/src/backend/commands/vacuumparallel.c +++ b/src/backend/commands/vacuumparallel.c @@ -261,8 +261,9 @@ struct ParallelVacuumState BufferAccessStrategy bstrategy; /* - * Error reporting state. The error callback is set only for workers - * processes during parallel index vacuum. + * Error reporting state. The error callback is set for the whole life of + * a worker process, and in the leader for as long as it processes indexes + * itself. */ char *relnamespace; char *relname; @@ -347,6 +348,12 @@ parallel_vacuum_init(Relation rel, Relation *indrels, int nindexes, pvs->will_parallel_vacuum = will_parallel_vacuum; pvs->bstrategy = bstrategy; pvs->heaprel = rel; + pvs->relnamespace = get_namespace_name(RelationGetNamespace(rel)); + pvs->relname = pstrdup(RelationGetRelationName(rel)); + + /* These fields will be filled during index vacuum or cleanup */ + pvs->indname = NULL; + pvs->status = PARALLEL_INDVAC_STATUS_INITIAL; EnterParallelMode(); pcxt = CreateParallelContext("postgres", "parallel_vacuum_main", @@ -538,6 +545,8 @@ parallel_vacuum_end(ParallelVacuumState *pvs, IndexBulkDeleteResult **istats) if (AmAutoVacuumWorkerProcess()) pv_shared_cost_params = NULL; + pfree(pvs->relnamespace); + pfree(pvs->relname); pfree(pvs->will_parallel_vacuum); pfree(pvs); } @@ -813,6 +822,7 @@ parallel_vacuum_process_all_indexes(ParallelVacuumState *pvs, int num_index_scan { int nworkers; PVIndVacStatus new_status; + ErrorContextCallback errcallback; Assert(!IsParallelWorker()); @@ -926,6 +936,15 @@ parallel_vacuum_process_all_indexes(ParallelVacuumState *pvs, int num_index_scan pvs->pcxt->nworkers_launched, nworkers))); } + /* + * Setup error traceback support for ereport() for as long as the leader + * processes indexes itself. + */ + errcallback.callback = parallel_vacuum_error_callback; + errcallback.arg = pvs; + errcallback.previous = error_context_stack; + error_context_stack = &errcallback; + /* Vacuum the indexes that can be processed by only leader process */ parallel_vacuum_process_unsafe_indexes(pvs); @@ -935,6 +954,9 @@ parallel_vacuum_process_all_indexes(ParallelVacuumState *pvs, int num_index_scan */ parallel_vacuum_process_safe_indexes(pvs); + /* Pop the error context stack */ + error_context_stack = errcallback.previous; + /* * Next, accumulate buffer and WAL usage. (This must wait for the workers * to finish, or we might get incomplete data.) diff --git a/src/test/modules/nbtree/Makefile b/src/test/modules/nbtree/Makefile index 20a1ca6a92b..1fcf3541ceb 100644 --- a/src/test/modules/nbtree/Makefile +++ b/src/test/modules/nbtree/Makefile @@ -3,7 +3,8 @@ EXTRA_INSTALL = src/test/modules/injection_points contrib/amcheck REGRESS = nbtree_half_dead_pages \ - nbtree_incomplete_splits + nbtree_incomplete_splits \ + nbtree_vacuum_error_context ISOLATION = backwards-scan-concurrent-splits \ predicate-empty-index diff --git a/src/test/modules/nbtree/expected/nbtree_vacuum_error_context.out b/src/test/modules/nbtree/expected/nbtree_vacuum_error_context.out new file mode 100644 index 00000000000..4ac5e39f4ad --- /dev/null +++ b/src/test/modules/nbtree/expected/nbtree_vacuum_error_context.out @@ -0,0 +1,68 @@ +-- +-- Test the error context reported while vacuum processes an index, in +-- particular by the leader of a parallel vacuum. +-- +set client_min_messages TO 'warning'; +create extension if not exists injection_points; +reset client_min_messages; +-- Wait until the deleted tuples are removable, so that vacuum really gets to +-- delete an index page. Same as in nbtree_half_dead_pages, which runs in the +-- same database. +CREATE OR REPLACE PROCEDURE wait_prunable() LANGUAGE plpgsql AS $$ + DECLARE + barrier xid8; + cutoff xid8; + BEGIN + barrier := pg_current_xact_id(); + LOOP + ROLLBACK; -- release MyProc->xmin, which could be the oldest + cutoff := removable_cutoff('pg_database'); + EXIT WHEN cutoff >= barrier; + PERFORM pg_sleep(.1); + END LOOP; + END +$$; +SELECT injection_points_set_local(); + injection_points_set_local +---------------------------- + +(1 row) + +create table nbtree_vacuum_error_context(id bigint, id2 bigint) + with (autovacuum_enabled = off); +insert into nbtree_vacuum_error_context + select g, g from generate_series(1, 30000) g; +-- Parallel vacuum needs more than one index +create index nbtree_vacuum_error_context_idx + on nbtree_vacuum_error_context (id); +create index nbtree_vacuum_error_context_idx2 + on nbtree_vacuum_error_context (id2); +-- Empty out whole leaf pages, so that vacuum deletes them. The range has to +-- be wide enough to cover a leaf page on a large block size build too. +delete from nbtree_vacuum_error_context where id > 10000 and id < 20000; +call wait_prunable(); +-- Error out in the middle of the index scan performed by vacuum +SELECT injection_points_attach('nbtree-leave-page-half-dead', 'error'); + injection_points_attach +------------------------- + +(1 row) + +VACUUM (PARALLEL 0) nbtree_vacuum_error_context; +ERROR: error triggered for injection point nbtree-leave-page-half-dead +CONTEXT: while vacuuming index "nbtree_vacuum_error_context_idx" of relation "public.nbtree_vacuum_error_context" +-- min_parallel_index_scan_size makes the small indexes eligible for parallel +-- vacuum, and max_parallel_workers leaves no worker to launch, so the leader +-- processes the indexes itself +SET min_parallel_index_scan_size = 0; +SET max_parallel_workers = 0; +VACUUM (PARALLEL 1) nbtree_vacuum_error_context; +ERROR: error triggered for injection point nbtree-leave-page-half-dead +CONTEXT: while vacuuming index "nbtree_vacuum_error_context_idx" of relation "public.nbtree_vacuum_error_context" +SELECT injection_points_detach('nbtree-leave-page-half-dead'); + injection_points_detach +------------------------- + +(1 row) + +drop extension injection_points; diff --git a/src/test/modules/nbtree/meson.build b/src/test/modules/nbtree/meson.build index b5dc026392e..30c01a3b0a6 100644 --- a/src/test/modules/nbtree/meson.build +++ b/src/test/modules/nbtree/meson.build @@ -12,6 +12,7 @@ tests += { 'sql': [ 'nbtree_half_dead_pages', 'nbtree_incomplete_splits', + 'nbtree_vacuum_error_context', ], }, 'isolation': { diff --git a/src/test/modules/nbtree/sql/nbtree_vacuum_error_context.sql b/src/test/modules/nbtree/sql/nbtree_vacuum_error_context.sql new file mode 100644 index 00000000000..aedb9f74216 --- /dev/null +++ b/src/test/modules/nbtree/sql/nbtree_vacuum_error_context.sql @@ -0,0 +1,59 @@ +-- +-- Test the error context reported while vacuum processes an index, in +-- particular by the leader of a parallel vacuum. +-- +set client_min_messages TO 'warning'; +create extension if not exists injection_points; +reset client_min_messages; + +-- Wait until the deleted tuples are removable, so that vacuum really gets to +-- delete an index page. Same as in nbtree_half_dead_pages, which runs in the +-- same database. +CREATE OR REPLACE PROCEDURE wait_prunable() LANGUAGE plpgsql AS $$ + DECLARE + barrier xid8; + cutoff xid8; + BEGIN + barrier := pg_current_xact_id(); + LOOP + ROLLBACK; -- release MyProc->xmin, which could be the oldest + cutoff := removable_cutoff('pg_database'); + EXIT WHEN cutoff >= barrier; + PERFORM pg_sleep(.1); + END LOOP; + END +$$; + +SELECT injection_points_set_local(); + +create table nbtree_vacuum_error_context(id bigint, id2 bigint) + with (autovacuum_enabled = off); + +insert into nbtree_vacuum_error_context + select g, g from generate_series(1, 30000) g; + +-- Parallel vacuum needs more than one index +create index nbtree_vacuum_error_context_idx + on nbtree_vacuum_error_context (id); +create index nbtree_vacuum_error_context_idx2 + on nbtree_vacuum_error_context (id2); + +-- Empty out whole leaf pages, so that vacuum deletes them. The range has to +-- be wide enough to cover a leaf page on a large block size build too. +delete from nbtree_vacuum_error_context where id > 10000 and id < 20000; +call wait_prunable(); + +-- Error out in the middle of the index scan performed by vacuum +SELECT injection_points_attach('nbtree-leave-page-half-dead', 'error'); + +VACUUM (PARALLEL 0) nbtree_vacuum_error_context; + +-- min_parallel_index_scan_size makes the small indexes eligible for parallel +-- vacuum, and max_parallel_workers leaves no worker to launch, so the leader +-- processes the indexes itself +SET min_parallel_index_scan_size = 0; +SET max_parallel_workers = 0; +VACUUM (PARALLEL 1) nbtree_vacuum_error_context; + +SELECT injection_points_detach('nbtree-leave-page-half-dead'); +drop extension injection_points; -- 2.47.3