From 2a614a1ea34d51f3b8edd237f1e46b3a23e6ee4f Mon Sep 17 00:00:00 2001 From: Vignesh C Date: Tue, 6 Oct 2026 16:20:56 +0530 Subject: [PATCH] Test to reproduce parallel apply error context issue. Test to reproduce parallel apply error context issue. --- src/backend/replication/logical/worker.c | 5 + .../t/101_parallel_apply_error_context.pl | 162 ++++++++++++++++++ 2 files changed, 167 insertions(+) create mode 100644 src/test/subscription/t/101_parallel_apply_error_context.pl diff --git a/src/backend/replication/logical/worker.c b/src/backend/replication/logical/worker.c index 08f9b1b514b..e245f22996b 100644 --- a/src/backend/replication/logical/worker.c +++ b/src/backend/replication/logical/worker.c @@ -2789,6 +2789,7 @@ apply_handle_begin(StringInfo s) break; case TRANS_PARALLEL_APPLY: + INJECTION_POINT("parallel-worker-before-xact-start", NULL); /* Hold the lock until the end of the transaction. */ pa_lock_transaction(MyParallelShared->xid, AccessExclusiveLock); pa_set_xact_state(MyParallelShared, PARALLEL_TRANS_STARTED); @@ -4504,6 +4505,10 @@ apply_handle_insert(StringInfo s) /* Set relation for error callback */ remote_ctx.rel = rel; + /* Test hook: pause here with remote_ctx pointing at this relation. */ + if (am_leader_apply_worker()) + INJECTION_POINT("leader-apply-insert-remote-ctx-set", NULL); + /* Check if the relation is safe for parallel apply */ check_relation_parallel_apply_safety(rel, LRPA_INSERT); diff --git a/src/test/subscription/t/101_parallel_apply_error_context.pl b/src/test/subscription/t/101_parallel_apply_error_context.pl new file mode 100644 index 00000000000..077aa824b64 --- /dev/null +++ b/src/test/subscription/t/101_parallel_apply_error_context.pl @@ -0,0 +1,162 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# Test that an error relayed from a parallel apply worker to the leader does +# not get a misleading extra CONTEXT line. This currently fails. +# +# ProcessParallelApplyMessage() (applyparallelworker.c) correctly captures +# the failing worker's own error context into edata.context, then does +# +# error_context_stack = apply_error_context_stack; +# ereport(ERROR, ..., errcontext("%s", edata.context)); +# +# But errfinish() unconditionally walks whatever error_context_stack is set +# to and invokes every registered callback, so the leader's own +# apply_error_callback() runs a second time and appends another CONTEXT +# line. That callback reads the single, file-static "remote_ctx" global, +# which by then describes whatever relation and transaction the LEADER +# itself happens to be applying directly at that moment -- not the +# transaction or relation that actually failed in the parallel worker. The +# relayed error ends up annotated with a bogus extra line blaming an +# unrelated, perfectly fine transaction. +# +# To freeze the leader while it is applying an unrelated, perfectly fine +# transaction, this test adds its own injection point +# (leader-apply-insert-remote-ctx-set, in apply_handle_insert() right after +# remote_ctx.rel is set) rather than a BEFORE INSERT trigger with +# pg_sleep(): any non-immutable trigger on a table makes +# logicalrep_rel_check_parallel_safety() mark it parallel-unsafe, and +# check_relation_parallel_apply_safety() then makes the leader wait for the +# last parallelized transaction to finish before even starting the insert +# -- which, with that transaction's worker deliberately frozen for this +# test, would simply deadlock. +use strict; +use warnings FATAL => 'all'; +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +if ($ENV{enable_injection_points} ne 'yes') +{ + plan skip_all => 'Injection points not supported by this build'; +} + +my $node_publisher = PostgreSQL::Test::Cluster->new('publisher'); +$node_publisher->init(allows_streaming => 'logical'); +$node_publisher->start; + +my $node_subscriber = PostgreSQL::Test::Cluster->new('subscriber'); +$node_subscriber->init; + +# Only one parallel apply worker, so the second transaction below finds no +# free worker and is applied directly by the leader itself. +$node_subscriber->append_conf('postgresql.conf', + "max_parallel_apply_workers_per_subscription = 1"); +$node_subscriber->start; +$node_subscriber->safe_psql('postgres', 'CREATE EXTENSION injection_points'); + +foreach my $node ($node_publisher, $node_subscriber) +{ + $node->safe_psql('postgres', qq( + CREATE TABLE tab_a (a int PRIMARY KEY); + CREATE TABLE tab_b (a int PRIMARY KEY); + )); +} + +# tab_a: the transaction that actually fails. Dispatched to the (only) +# parallel worker. +$node_subscriber->safe_psql('postgres', qq[ + CREATE OR REPLACE FUNCTION tab_a_boom_fn() + RETURNS trigger + LANGUAGE plpgsql + AS \$\$ + BEGIN + RAISE EXCEPTION 'tab_a trigger boom'; + END; + \$\$; + + CREATE TRIGGER tab_a_boom_tg + BEFORE INSERT ON tab_a + FOR EACH ROW + EXECUTE FUNCTION tab_a_boom_fn(); +]); +$node_subscriber->safe_psql('postgres', + "ALTER TABLE tab_a ENABLE REPLICA TRIGGER tab_a_boom_tg;"); + +$node_publisher->safe_psql('postgres', + 'CREATE PUBLICATION pub FOR TABLE tab_a, tab_b'); + +my $publisher_connstr = $node_publisher->connstr . ' dbname=postgres'; +$node_subscriber->safe_psql('postgres', + "CREATE SUBSCRIPTION sub CONNECTION '$publisher_connstr' PUBLICATION pub" +); +$node_subscriber->wait_for_subscription_sync($node_publisher, 'sub'); + +# Freeze the (only) parallel apply worker right after it is assigned the +# tab_a transaction, before it drains anything beyond the BEGIN message. +$node_subscriber->safe_psql('postgres', + "SELECT injection_points_attach('parallel-worker-before-xact-start', 'wait')" +); + +# Freeze the leader itself right after it sets remote_ctx to describe +# tab_b, while actually applying tab_b's insert directly (see the note at +# the top of this file on why this isn't a trigger-based sleep). +$node_subscriber->safe_psql('postgres', + "SELECT injection_points_attach('leader-apply-insert-remote-ctx-set', 'wait')" +); + +my $log_offset = -s $node_subscriber->logfile; + +# TX1: dispatched to the only parallel worker, which immediately freezes +# before touching tab_a. +$node_publisher->safe_psql('postgres', 'INSERT INTO tab_a VALUES (1)'); +$node_subscriber->wait_for_event('logical replication parallel worker', + 'parallel-worker-before-xact-start'); + +# TX2: an unrelated, perfectly fine transaction. No parallel worker is free +# (the only one is frozen above), so the leader applies it directly, and +# freezes right after remote_ctx is set to describe tab_b. +$node_publisher->safe_psql('postgres', 'INSERT INTO tab_b VALUES (1)'); +$node_subscriber->wait_for_event('logical replication apply worker', + 'leader-apply-insert-remote-ctx-set'); + +# While the leader is frozen there -- i.e. remote_ctx still describes +# tab_b/TX2 -- let the frozen worker proceed. It will run tab_a's trigger, +# raise an error, and relay it to the leader asynchronously; the leader's +# injection-point wait loop still calls CHECK_FOR_INTERRUPTS(), so it picks +# the relayed error up without needing to be woken first. +$node_subscriber->safe_psql('postgres', + "SELECT injection_points_wakeup('parallel-worker-before-xact-start')"); + +$node_subscriber->wait_for_log( + qr/logical replication parallel apply worker exited due to error/, + $log_offset); + +my $log_contents = slurp_file($node_subscriber->logfile, $log_offset); + +# The error did originate from the tab_a/TX1 worker ... +like( + $log_contents, + qr/tab_a trigger boom/, + 'the relayed error is the tab_a trigger exception' +); + +# ... and it must not be polluted with an unrelated CONTEXT line describing +# whatever the leader itself happened to be doing at that moment (tab_b): +# the leader's own context-stack walk in ProcessParallelApplyMessage() +# currently appends exactly such a bogus line, so this fails today. +unlike( + $log_contents, + qr/relation "public\.tab_b"/, + 'the relayed error must not be annotated with the leader\'s own, ' + . 'unrelated tab_b activity as additional context' +); + +$node_subscriber->safe_psql('postgres', + "SELECT injection_points_detach('parallel-worker-before-xact-start')"); +$node_subscriber->safe_psql('postgres', + "SELECT injection_points_detach('leader-apply-insert-remote-ctx-set')"); + +$node_subscriber->stop('immediate'); +$node_publisher->stop('immediate'); + +done_testing(); -- 2.55.0