From 9b6e8e31dedd16159d60df6a1364176523a0499d Mon Sep 17 00:00:00 2001 From: Chee Wooson Date: Mon, 28 Sep 2026 17:34:54 +0800 Subject: [PATCH v3 3/3] Test abort-window conflicts and crashed multixact updaters Use an injection point after the abort status write to verify that conflicting requests wait for ProcArray cleanup, including DELETE and key-changing UPDATE, and check the final row contents for these two cases. Also crash an updater while a prepared key-share locker survives, then verify that a repeatable-read update can proceed after recovery. --- src/backend/access/transam/xact.c | 6 ++ src/test/modules/injection_points/Makefile | 3 +- .../expected/multixact-aborted-updater.out | 94 ++++++++++++++++++++++++++++++++ src/test/modules/injection_points/meson.build | 1 + .../specs/multixact-aborted-updater.spec | 104 +++++++++++++++++++++++++++++++++++ src/test/recovery/meson.build | 1 + .../t/058_multixact_crashed_updater.pl | 59 ++++++++++++++++++++ 7 files changed, 267 insertions(+), 1 deletion(-) create mode 100644 src/test/modules/injection_points/expected/multixact-aborted-updater.out create mode 100644 src/test/modules/injection_points/specs/multixact-aborted-updater.spec create mode 100644 src/test/recovery/t/058_multixact_crashed_updater.pl diff --git a/src/backend/access/transam/xact.c b/src/backend/access/transam/xact.c index 7b67db514ec..ffa64da35f8 100644 --- a/src/backend/access/transam/xact.c +++ b/src/backend/access/transam/xact.c @@ -1925,6 +1925,12 @@ RecordTransactionAbort(bool isSubXact) if (ndroppedstats) pfree(droppedstats); + /* + * Test the window where the transaction is aborted in pg_xact but still + * present in ProcArray. + */ + INJECTION_POINT("transaction-abort-after-clog", NULL); + return latestXid; } diff --git a/src/test/modules/injection_points/Makefile b/src/test/modules/injection_points/Makefile index 136f0f77951..74eb91a8cff 100644 --- a/src/test/modules/injection_points/Makefile +++ b/src/test/modules/injection_points/Makefile @@ -29,7 +29,8 @@ ISOLATION = basic \ syscache-update-pruned \ wait_cleanup \ heap_lock_update \ - on_conflict_probe_window + on_conflict_probe_window \ + multixact-aborted-updater # some isolation tests require wal_level=replica ISOLATION_OPTS = --temp-config $(top_srcdir)/src/test/modules/injection_points/extra.conf diff --git a/src/test/modules/injection_points/expected/multixact-aborted-updater.out b/src/test/modules/injection_points/expected/multixact-aborted-updater.out new file mode 100644 index 00000000000..baf8991949f --- /dev/null +++ b/src/test/modules/injection_points/expected/multixact-aborted-updater.out @@ -0,0 +1,94 @@ +Parsed test spec with 4 sessions + +starting permutation: s1begin s1update s2lock s1abort s3nowait wake +step s1begin: BEGIN; +step s1update: UPDATE mxact_abort SET filler = 's1' WHERE id = 1; +step s2lock: SELECT * FROM mxact_abort WHERE id = 1 FOR KEY SHARE; +id|filler +--+------- + 1|initial +(1 row) + +step s1abort: ROLLBACK; +step s3nowait: SELECT * FROM mxact_abort WHERE id = 1 FOR NO KEY UPDATE NOWAIT; +ERROR: could not obtain lock on row in relation "mxact_abort" +step wake: + SELECT FROM injection_points_detach('transaction-abort-after-clog'); + SELECT FROM injection_points_wakeup('transaction-abort-after-clog'); + +step s1abort: <... completed> + +starting permutation: s1begin s1update s2lock s1abort s3update wake +step s1begin: BEGIN; +step s1update: UPDATE mxact_abort SET filler = 's1' WHERE id = 1; +step s2lock: SELECT * FROM mxact_abort WHERE id = 1 FOR KEY SHARE; +id|filler +--+------- + 1|initial +(1 row) + +step s1abort: ROLLBACK; +step s3update: + DO $$ + BEGIN + UPDATE mxact_abort SET filler = 's3' WHERE id = 1; + RAISE NOTICE 'session 3 update succeeded'; + EXCEPTION WHEN OTHERS THEN + RAISE NOTICE 'session 3 update failed: %', split_part(SQLERRM, ':', 1); + END + $$; + +step wake: + SELECT FROM injection_points_detach('transaction-abort-after-clog'); + SELECT FROM injection_points_wakeup('transaction-abort-after-clog'); + +s3: NOTICE: session 3 update succeeded +step s3update: <... completed> +step s1abort: <... completed> + +starting permutation: s1begin s1update s2lock s1abort s3delete wake s3check +step s1begin: BEGIN; +step s1update: UPDATE mxact_abort SET filler = 's1' WHERE id = 1; +step s2lock: SELECT * FROM mxact_abort WHERE id = 1 FOR KEY SHARE; +id|filler +--+------- + 1|initial +(1 row) + +step s1abort: ROLLBACK; +step s3delete: DELETE FROM mxact_abort WHERE id = 1; +step wake: + SELECT FROM injection_points_detach('transaction-abort-after-clog'); + SELECT FROM injection_points_wakeup('transaction-abort-after-clog'); + +step s3delete: <... completed> +step s1abort: <... completed> +step s3check: SELECT * FROM mxact_abort ORDER BY id; +id|filler +--+------ +(0 rows) + + +starting permutation: s1begin s1update s2lock s1abort s3keyupdate wake s3check +step s1begin: BEGIN; +step s1update: UPDATE mxact_abort SET filler = 's1' WHERE id = 1; +step s2lock: SELECT * FROM mxact_abort WHERE id = 1 FOR KEY SHARE; +id|filler +--+------- + 1|initial +(1 row) + +step s1abort: ROLLBACK; +step s3keyupdate: UPDATE mxact_abort SET id = 2 WHERE id = 1; +step wake: + SELECT FROM injection_points_detach('transaction-abort-after-clog'); + SELECT FROM injection_points_wakeup('transaction-abort-after-clog'); + +step s3keyupdate: <... completed> +step s1abort: <... completed> +step s3check: SELECT * FROM mxact_abort ORDER BY id; +id|filler +--+------- + 2|initial +(1 row) + diff --git a/src/test/modules/injection_points/meson.build b/src/test/modules/injection_points/meson.build index db6b93a8115..a161b385133 100644 --- a/src/test/modules/injection_points/meson.build +++ b/src/test/modules/injection_points/meson.build @@ -59,6 +59,7 @@ tests += { 'wait_cleanup', 'heap_lock_update', 'on_conflict_probe_window', + 'multixact-aborted-updater', ], 'runningcheck': false, # see syscache-update-pruned # Some tests wait for all snapshots, so avoid parallel execution diff --git a/src/test/modules/injection_points/specs/multixact-aborted-updater.spec b/src/test/modules/injection_points/specs/multixact-aborted-updater.spec new file mode 100644 index 00000000000..c91cabdca7c --- /dev/null +++ b/src/test/modules/injection_points/specs/multixact-aborted-updater.spec @@ -0,0 +1,104 @@ +# Test concurrent behavior during an updater's abort window (between the +# pg_xact write and ProcArray cleanup) and after it closes. +# +# Session 1 pauses after recording its abort in pg_xact, before ProcArray +# cleanup. During the window the updater still reports running, so a +# conflicting NOWAIT lock request fails, and UPDATE and DELETE wait out +# the window. Once the abort finishes, these operations succeed. A new +# UPDATE must not retain the aborted updater when expanding the multixact. + +setup +{ + CREATE EXTENSION injection_points; + + CREATE TABLE mxact_abort (id int PRIMARY KEY, filler text); + INSERT INTO mxact_abort VALUES (1, 'initial'); +} + +teardown +{ + DROP TABLE mxact_abort; + DROP EXTENSION injection_points; +} + +session s1 +setup { + SELECT FROM injection_points_set_local(); + SELECT FROM injection_points_attach('transaction-abort-after-clog', 'wait'); +} +step s1begin { BEGIN; } +step s1update { UPDATE mxact_abort SET filler = 's1' WHERE id = 1; } +step s1abort { ROLLBACK; } + +session s2 +# The compatible key-share lock makes xmax a multixact with session 1 as updater. +step s2lock { SELECT * FROM mxact_abort WHERE id = 1 FOR KEY SHARE; } + +session s3 +step s3nowait { SELECT * FROM mxact_abort WHERE id = 1 FOR NO KEY UPDATE NOWAIT; } +# Omit variable XIDs from the error message. +step s3update { + DO $$ + BEGIN + UPDATE mxact_abort SET filler = 's3' WHERE id = 1; + RAISE NOTICE 'session 3 update succeeded'; + EXCEPTION WHEN OTHERS THEN + RAISE NOTICE 'session 3 update failed: %', split_part(SQLERRM, ':', 1); + END + $$; +} +step s3delete { DELETE FROM mxact_abort WHERE id = 1; } +step s3keyupdate { UPDATE mxact_abort SET id = 2 WHERE id = 1; } +step s3check { SELECT * FROM mxact_abort ORDER BY id; } + +session s4 +step wake { + SELECT FROM injection_points_detach('transaction-abort-after-clog'); + SELECT FROM injection_points_wakeup('transaction-abort-after-clog'); +} + +# During the window the updater, though already aborted in pg_xact, still +# reports running: the conflicting NOWAIT request fails. +permutation + s1begin + s1update + s2lock + # Pause after recording the abort, before ProcArray cleanup. + s1abort + s3nowait + wake + +# An ordinary UPDATE waits out the window instead of proceeding; when the +# abort completes, it succeeds; the updater is no longer running when +# multixact expansion examines its members. +# Report the abort after the conflicting operation in these permutations +# to keep the completion output stable. +permutation + s1begin + s1update + s2lock + s1abort(s3update) + s3update + wake + +# DELETE must wait rather than report TM_Updated and fail the executor's +# traversed assertion when it re-locks the original tuple. +permutation + s1begin + s1update + s2lock + s1abort(s3delete) + s3delete + wake + s3check + +# A key-changing UPDATE must also wait before deciding which members of +# the multixact survive. The new key must retain the original row's value. +permutation + s1begin + s1update + s2lock + s1abort(s3keyupdate) + s3keyupdate + wake + s3check diff --git a/src/test/recovery/meson.build b/src/test/recovery/meson.build index ebb12dd8766..9f30ca2f884 100644 --- a/src/test/recovery/meson.build +++ b/src/test/recovery/meson.build @@ -66,6 +66,7 @@ tests += { 't/055_cascade_reconnect.pl', 't/056_standby_snapshot_export.pl', 't/057_snapshot_commit_race.pl', + 't/058_multixact_crashed_updater.pl', ], }, } diff --git a/src/test/recovery/t/058_multixact_crashed_updater.pl b/src/test/recovery/t/058_multixact_crashed_updater.pl new file mode 100644 index 00000000000..036baf7bd0b --- /dev/null +++ b/src/test/recovery/t/058_multixact_crashed_updater.pl @@ -0,0 +1,59 @@ +# Copyright (c) 2026, PostgreSQL Global Development Group + +# A crashed updater can remain in a tuple's multixact alongside a prepared +# key-share locker. The updater never committed, so a later non-key update +# must be able to proceed, including at REPEATABLE READ isolation level. + +use strict; +use warnings FATAL => 'all'; + +use PostgreSQL::Test::Cluster; +use PostgreSQL::Test::Utils; +use Test::More; + +my $node = PostgreSQL::Test::Cluster->new('main'); +$node->init; +$node->append_conf('postgresql.conf', 'max_prepared_transactions = 10'); +$node->start; + +$node->safe_psql('postgres', q( + CREATE TABLE mx_crashed (id int PRIMARY KEY, payload text); + INSERT INTO mx_crashed VALUES (1, 'initial'); +)); + +# Keep the updater open while another transaction adds a compatible locker. +# PREPARE also flushes the preceding heap update and multixact WAL records. +my $updater = $node->background_psql('postgres'); +$updater->query_safe('BEGIN'); +$updater->query_safe(q(UPDATE mx_crashed SET payload = 'crashed' WHERE id = 1)); + +$node->safe_psql('postgres', q( + BEGIN; + SELECT id FROM mx_crashed WHERE id = 1 FOR KEY SHARE; + PREPARE TRANSACTION 'mx_crashed_locker'; +)); + +$node->stop('immediate'); +$updater->{run}->finish; +$node->start; + +is($node->safe_psql('postgres', + q(SELECT count(*) FROM pg_prepared_xacts WHERE gid = 'mx_crashed_locker')), + '1', 'key-share locker survived crash recovery'); + +my ($stdout, $stderr); +my $ret = $node->psql('postgres', q( + BEGIN ISOLATION LEVEL REPEATABLE READ; + UPDATE mx_crashed SET payload = 'after crash' WHERE id = 1; + COMMIT; +), stdout => \$stdout, stderr => \$stderr); +is($ret, 0, 'repeatable read update ignores the crashed updater') + or diag($stderr); + +is($node->safe_psql('postgres', + q(SELECT payload FROM mx_crashed WHERE id = 1)), + 'after crash', 'row was updated'); + +$node->safe_psql('postgres', q(ROLLBACK PREPARED 'mx_crashed_locker')); + +done_testing(); -- 2.43.0