Hi,

AI review identified an inconsistency in parallel vacuum (8e1fae1938).
I checked it myself and it is there on HEAD and on every branch since
PG15. Patch with a test attached. I don't think back-patching is
necessary as it's not a bug, only the context that gets reported.

The leader processes some of the indexes itself, through the same
function as the workers, parallel_vacuum_process_one_index(), and that
function records the index name for the error context, but only the
workers install the callback that reads it. The leader keeps the lazy
vacuum callback with whatever the heap scan last left there, so an
error it raises while vacuuming or cleaning up an index names the
table: "while scanning relation" for the index pass after the heap
scan and "while scanning block N of relation" for a pass in the middle
of it, where it should read "while vacuuming index ... of relation
...". Errors relayed from the workers get the same stale line
appended. Before 8e1fae1938 the leader went through
lazy_vacuum_one_index() and lazy_cleanup_one_index(), which set the
phase and the index name.

With PG19 this shows up in every autovacuum log once parallel
autovacuum is turned on, and that log line is often the only evidence
there is about which index a bad page belongs to.

The fix installs the parallel vacuum error callback in the leader for
as long as it processes indexes itself, and sets the heap scan phase
aside for that time so the stale line is not printed alongside it.

--
Bharath Rupireddy
Amazon Web Services: https://aws.amazon.com
From 4590b63c7c88ed2372dc5be2be02227e47841161 Mon Sep 17 00:00:00 2001
From: Bharath Rupireddy <[email protected]>
Date: Sun, 27 Sep 2026 00:07:24 +0000
Subject: [PATCH v1] Fix leader's error context during parallel index
 vacuuming.

Commit 8e1fae1938 moved the parallel vacuum code to
vacuumparallel.c, and since then the leader processes indexes
through parallel_vacuum_process_one_index() just like the workers
do. Only the workers, however, install the error context callback
that reports the index being worked on.

The leader keeps the lazy vacuum error callback with the phase
left over from the heap scan, so an error it raises while
vacuuming or cleaning up an index reports the heap instead of the
index, as "while scanning block N of relation" for an index pass
in the middle of the scan or as "while scanning relation" for the
pass after it. Errors relayed from the workers get the same stale
line appended. Before 8e1fae1938 the leader went through
lazy_vacuum_one_index() and lazy_cleanup_one_index(), which set
the phase and the index name. The log is often the only evidence
there is when an index page turns out to be bad, and with parallel
autovacuum in PG19 every autovacuum of a large table goes through
this path.

Fix this by installing the parallel vacuum error callback in the
leader for as long as it processes indexes itself, and by filling
in the relation names it needs when the parallel state is created.
The callers set the lazy vacuum phase aside for the duration of
the parallel index phases so that the heap scan line is not
reported alongside it.

This is an inconsistency in the context that gets reported rather
than a bug, so it is not back-patched.

Oversight in commit 8e1fae1938.

Reported-by: Claude Code
Author: Bharath Rupireddy <[email protected]>
Discussion: https://postgr.es/m/<<message-id>>
---
 src/backend/access/heap/vacuumlazy.c          | 23 ++++++-
 src/backend/commands/vacuumparallel.c         | 26 ++++++-
 src/test/modules/nbtree/Makefile              |  3 +-
 .../expected/nbtree_vacuum_error_context.out  | 68 +++++++++++++++++++
 src/test/modules/nbtree/meson.build           |  1 +
 .../sql/nbtree_vacuum_error_context.sql       | 59 ++++++++++++++++
 6 files changed, 175 insertions(+), 5 deletions(-)
 create mode 100644 src/test/modules/nbtree/expected/nbtree_vacuum_error_context.out
 create mode 100644 src/test/modules/nbtree/sql/nbtree_vacuum_error_context.sql

diff --git a/src/backend/access/heap/vacuumlazy.c b/src/backend/access/heap/vacuumlazy.c
index 997d84a77b3..f5a11bac026 100644
--- a/src/backend/access/heap/vacuumlazy.c
+++ b/src/backend/access/heap/vacuumlazy.c
@@ -2564,10 +2564,20 @@ lazy_vacuum_all_indexes(LVRelState *vacrel)
 	}
 	else
 	{
-		/* Outsource everything to parallel variant */
+		LVSavedErrInfo saved_err_info;
+
+		/*
+		 * Outsource everything to parallel variant. An error raised while the
+		 * leader itself processes an index gets its context from the parallel
+		 * vacuum code, so stop reporting the heap phase we came from.
+		 */
+		update_vacuum_error_info(vacrel, &saved_err_info,
+								 VACUUM_ERRCB_PHASE_UNKNOWN,
+								 InvalidBlockNumber, InvalidOffsetNumber);
 		parallel_vacuum_bulkdel_all_indexes(vacrel->pvs, old_live_tuples,
 											vacrel->num_index_scans,
 											&(vacrel->worker_usage.vacuum));
+		restore_vacuum_error_info(vacrel, &saved_err_info);
 
 		/*
 		 * Do a postcheck to consider applying wraparound failsafe now.  Note
@@ -3008,11 +3018,20 @@ lazy_cleanup_all_indexes(LVRelState *vacrel)
 	}
 	else
 	{
-		/* Outsource everything to parallel variant */
+		LVSavedErrInfo saved_err_info;
+
+		/*
+		 * Outsource everything to parallel variant. See
+		 * lazy_vacuum_all_indexes() for the error context handling.
+		 */
+		update_vacuum_error_info(vacrel, &saved_err_info,
+								 VACUUM_ERRCB_PHASE_UNKNOWN,
+								 InvalidBlockNumber, InvalidOffsetNumber);
 		parallel_vacuum_cleanup_all_indexes(vacrel->pvs, reltuples,
 											vacrel->num_index_scans,
 											estimated_count,
 											&(vacrel->worker_usage.cleanup));
+		restore_vacuum_error_info(vacrel, &saved_err_info);
 	}
 
 	/* Reset the progress counters */
diff --git a/src/backend/commands/vacuumparallel.c b/src/backend/commands/vacuumparallel.c
index 4532da60c84..f657494a496 100644
--- a/src/backend/commands/vacuumparallel.c
+++ b/src/backend/commands/vacuumparallel.c
@@ -261,8 +261,9 @@ struct ParallelVacuumState
 	BufferAccessStrategy bstrategy;
 
 	/*
-	 * Error reporting state.  The error callback is set only for workers
-	 * processes during parallel index vacuum.
+	 * Error reporting state. The error callback is set for the whole life of
+	 * a worker process, and in the leader for as long as it processes indexes
+	 * itself.
 	 */
 	char	   *relnamespace;
 	char	   *relname;
@@ -347,6 +348,12 @@ parallel_vacuum_init(Relation rel, Relation *indrels, int nindexes,
 	pvs->will_parallel_vacuum = will_parallel_vacuum;
 	pvs->bstrategy = bstrategy;
 	pvs->heaprel = rel;
+	pvs->relnamespace = get_namespace_name(RelationGetNamespace(rel));
+	pvs->relname = pstrdup(RelationGetRelationName(rel));
+
+	/* These fields will be filled during index vacuum or cleanup */
+	pvs->indname = NULL;
+	pvs->status = PARALLEL_INDVAC_STATUS_INITIAL;
 
 	EnterParallelMode();
 	pcxt = CreateParallelContext("postgres", "parallel_vacuum_main",
@@ -538,6 +545,8 @@ parallel_vacuum_end(ParallelVacuumState *pvs, IndexBulkDeleteResult **istats)
 	if (AmAutoVacuumWorkerProcess())
 		pv_shared_cost_params = NULL;
 
+	pfree(pvs->relnamespace);
+	pfree(pvs->relname);
 	pfree(pvs->will_parallel_vacuum);
 	pfree(pvs);
 }
@@ -813,6 +822,7 @@ parallel_vacuum_process_all_indexes(ParallelVacuumState *pvs, int num_index_scan
 {
 	int			nworkers;
 	PVIndVacStatus new_status;
+	ErrorContextCallback errcallback;
 
 	Assert(!IsParallelWorker());
 
@@ -926,6 +936,15 @@ parallel_vacuum_process_all_indexes(ParallelVacuumState *pvs, int num_index_scan
 							pvs->pcxt->nworkers_launched, nworkers)));
 	}
 
+	/*
+	 * Setup error traceback support for ereport() for as long as the leader
+	 * processes indexes itself.
+	 */
+	errcallback.callback = parallel_vacuum_error_callback;
+	errcallback.arg = pvs;
+	errcallback.previous = error_context_stack;
+	error_context_stack = &errcallback;
+
 	/* Vacuum the indexes that can be processed by only leader process */
 	parallel_vacuum_process_unsafe_indexes(pvs);
 
@@ -935,6 +954,9 @@ parallel_vacuum_process_all_indexes(ParallelVacuumState *pvs, int num_index_scan
 	 */
 	parallel_vacuum_process_safe_indexes(pvs);
 
+	/* Pop the error context stack */
+	error_context_stack = errcallback.previous;
+
 	/*
 	 * Next, accumulate buffer and WAL usage.  (This must wait for the workers
 	 * to finish, or we might get incomplete data.)
diff --git a/src/test/modules/nbtree/Makefile b/src/test/modules/nbtree/Makefile
index 20a1ca6a92b..1fcf3541ceb 100644
--- a/src/test/modules/nbtree/Makefile
+++ b/src/test/modules/nbtree/Makefile
@@ -3,7 +3,8 @@
 EXTRA_INSTALL = src/test/modules/injection_points contrib/amcheck
 
 REGRESS = nbtree_half_dead_pages \
-	nbtree_incomplete_splits
+	nbtree_incomplete_splits \
+	nbtree_vacuum_error_context
 
 ISOLATION = backwards-scan-concurrent-splits \
 	predicate-empty-index
diff --git a/src/test/modules/nbtree/expected/nbtree_vacuum_error_context.out b/src/test/modules/nbtree/expected/nbtree_vacuum_error_context.out
new file mode 100644
index 00000000000..4ac5e39f4ad
--- /dev/null
+++ b/src/test/modules/nbtree/expected/nbtree_vacuum_error_context.out
@@ -0,0 +1,68 @@
+--
+-- Test the error context reported while vacuum processes an index, in
+-- particular by the leader of a parallel vacuum.
+--
+set client_min_messages TO 'warning';
+create extension if not exists injection_points;
+reset client_min_messages;
+-- Wait until the deleted tuples are removable, so that vacuum really gets to
+-- delete an index page. Same as in nbtree_half_dead_pages, which runs in the
+-- same database.
+CREATE OR REPLACE PROCEDURE wait_prunable() LANGUAGE plpgsql AS $$
+	DECLARE
+		barrier xid8;
+		cutoff xid8;
+	BEGIN
+		barrier := pg_current_xact_id();
+		LOOP
+			ROLLBACK;  -- release MyProc->xmin, which could be the oldest
+			cutoff := removable_cutoff('pg_database');
+			EXIT WHEN cutoff >= barrier;
+			PERFORM pg_sleep(.1);
+		END LOOP;
+	END
+$$;
+SELECT injection_points_set_local();
+ injection_points_set_local 
+----------------------------
+ 
+(1 row)
+
+create table nbtree_vacuum_error_context(id bigint, id2 bigint)
+  with (autovacuum_enabled = off);
+insert into nbtree_vacuum_error_context
+  select g, g from generate_series(1, 30000) g;
+-- Parallel vacuum needs more than one index
+create index nbtree_vacuum_error_context_idx
+  on nbtree_vacuum_error_context (id);
+create index nbtree_vacuum_error_context_idx2
+  on nbtree_vacuum_error_context (id2);
+-- Empty out whole leaf pages, so that vacuum deletes them. The range has to
+-- be wide enough to cover a leaf page on a large block size build too.
+delete from nbtree_vacuum_error_context where id > 10000 and id < 20000;
+call wait_prunable();
+-- Error out in the middle of the index scan performed by vacuum
+SELECT injection_points_attach('nbtree-leave-page-half-dead', 'error');
+ injection_points_attach 
+-------------------------
+ 
+(1 row)
+
+VACUUM (PARALLEL 0) nbtree_vacuum_error_context;
+ERROR:  error triggered for injection point nbtree-leave-page-half-dead
+CONTEXT:  while vacuuming index "nbtree_vacuum_error_context_idx" of relation "public.nbtree_vacuum_error_context"
+-- min_parallel_index_scan_size makes the small indexes eligible for parallel
+-- vacuum, and max_parallel_workers leaves no worker to launch, so the leader
+-- processes the indexes itself
+SET min_parallel_index_scan_size = 0;
+SET max_parallel_workers = 0;
+VACUUM (PARALLEL 1) nbtree_vacuum_error_context;
+ERROR:  error triggered for injection point nbtree-leave-page-half-dead
+CONTEXT:  while vacuuming index "nbtree_vacuum_error_context_idx" of relation "public.nbtree_vacuum_error_context"
+SELECT injection_points_detach('nbtree-leave-page-half-dead');
+ injection_points_detach 
+-------------------------
+ 
+(1 row)
+
+drop extension injection_points;
diff --git a/src/test/modules/nbtree/meson.build b/src/test/modules/nbtree/meson.build
index b5dc026392e..30c01a3b0a6 100644
--- a/src/test/modules/nbtree/meson.build
+++ b/src/test/modules/nbtree/meson.build
@@ -12,6 +12,7 @@ tests += {
     'sql': [
       'nbtree_half_dead_pages',
       'nbtree_incomplete_splits',
+      'nbtree_vacuum_error_context',
     ],
   },
   'isolation': {
diff --git a/src/test/modules/nbtree/sql/nbtree_vacuum_error_context.sql b/src/test/modules/nbtree/sql/nbtree_vacuum_error_context.sql
new file mode 100644
index 00000000000..aedb9f74216
--- /dev/null
+++ b/src/test/modules/nbtree/sql/nbtree_vacuum_error_context.sql
@@ -0,0 +1,59 @@
+--
+-- Test the error context reported while vacuum processes an index, in
+-- particular by the leader of a parallel vacuum.
+--
+set client_min_messages TO 'warning';
+create extension if not exists injection_points;
+reset client_min_messages;
+
+-- Wait until the deleted tuples are removable, so that vacuum really gets to
+-- delete an index page. Same as in nbtree_half_dead_pages, which runs in the
+-- same database.
+CREATE OR REPLACE PROCEDURE wait_prunable() LANGUAGE plpgsql AS $$
+	DECLARE
+		barrier xid8;
+		cutoff xid8;
+	BEGIN
+		barrier := pg_current_xact_id();
+		LOOP
+			ROLLBACK;  -- release MyProc->xmin, which could be the oldest
+			cutoff := removable_cutoff('pg_database');
+			EXIT WHEN cutoff >= barrier;
+			PERFORM pg_sleep(.1);
+		END LOOP;
+	END
+$$;
+
+SELECT injection_points_set_local();
+
+create table nbtree_vacuum_error_context(id bigint, id2 bigint)
+  with (autovacuum_enabled = off);
+
+insert into nbtree_vacuum_error_context
+  select g, g from generate_series(1, 30000) g;
+
+-- Parallel vacuum needs more than one index
+create index nbtree_vacuum_error_context_idx
+  on nbtree_vacuum_error_context (id);
+create index nbtree_vacuum_error_context_idx2
+  on nbtree_vacuum_error_context (id2);
+
+-- Empty out whole leaf pages, so that vacuum deletes them. The range has to
+-- be wide enough to cover a leaf page on a large block size build too.
+delete from nbtree_vacuum_error_context where id > 10000 and id < 20000;
+call wait_prunable();
+
+-- Error out in the middle of the index scan performed by vacuum
+SELECT injection_points_attach('nbtree-leave-page-half-dead', 'error');
+
+VACUUM (PARALLEL 0) nbtree_vacuum_error_context;
+
+-- min_parallel_index_scan_size makes the small indexes eligible for parallel
+-- vacuum, and max_parallel_workers leaves no worker to launch, so the leader
+-- processes the indexes itself
+SET min_parallel_index_scan_size = 0;
+SET max_parallel_workers = 0;
+VACUUM (PARALLEL 1) nbtree_vacuum_error_context;
+
+SELECT injection_points_detach('nbtree-leave-page-half-dead');
+drop extension injection_points;
-- 
2.47.3

Reply via email to