From 31077001143c5a580759c9f82231ee9a532c9aa7 Mon Sep 17 00:00:00 2001 From: shihao zhong Date: Sun, 4 Oct 2026 23:30:37 -0400 Subject: [PATCH v1 2/2] Add row estimate tests for IS [NOT] DISTINCT FROM Check that join estimates count rows with NULL on both sides. Discussion: https://postgr.es/m/17545-a0ca4de888953169@postgresql.org --- src/test/regress/expected/planner_est.out | 36 +++++++++++++++++++++++ src/test/regress/sql/planner_est.sql | 25 ++++++++++++++++ 2 files changed, 61 insertions(+) diff --git a/src/test/regress/expected/planner_est.out b/src/test/regress/expected/planner_est.out index 236cb274a78..07b88e5229d 100644 --- a/src/test/regress/expected/planner_est.out +++ b/src/test/regress/expected/planner_est.out @@ -221,4 +221,40 @@ EXPLAIN (COSTS OFF) SELECT * FROM char_table_1 WHERE c < 'Q'; Filter: (c < 'Q'::"char") (2 rows) +-- +-- Test IS [NOT] DISTINCT FROM join estimates with NULLs on both sides. Only +-- the top plan line is shown, since the rest is not stable across platforms. +-- +CREATE TEMP TABLE distinct_t1 AS + SELECT CASE WHEN i > 30 THEN i END AS a FROM generate_series(1, 100) i; +CREATE TEMP TABLE distinct_t2 AS + SELECT CASE WHEN i > 100 THEN i - 100 END AS a FROM generate_series(1, 200) i; +ANALYZE distinct_t1, distinct_t2; +-- Ensure pairs of NULLs are counted +SELECT * FROM explain_mask_costs($$ +SELECT * FROM distinct_t1 t1 JOIN distinct_t2 t2 ON t1.a IS NOT DISTINCT FROM t2.a;$$, +true, true, false, true) LIMIT 1; + explain_mask_costs +-------------------------------------------------------------------------- + Nested Loop (cost=N..N rows=3070 width=N) (actual rows=3070.00 loops=1) +(1 row) + +-- The equivalent OR clause should get about the same estimate +SELECT * FROM explain_mask_costs($$ +SELECT * FROM distinct_t1 t1 JOIN distinct_t2 t2 ON t1.a = t2.a OR (t1.a IS NULL AND t2.a IS NULL);$$, +true, true, false, true) LIMIT 1; + explain_mask_costs +-------------------------------------------------------------------------- + Nested Loop (cost=N..N rows=3060 width=N) (actual rows=3070.00 loops=1) +(1 row) + +-- Ensure pairs of NULLs are not counted +SELECT * FROM explain_mask_costs($$ +SELECT * FROM distinct_t1 t1 JOIN distinct_t2 t2 ON t1.a IS DISTINCT FROM t2.a;$$, +true, true, false, true) LIMIT 1; + explain_mask_costs +---------------------------------------------------------------------------- + Nested Loop (cost=N..N rows=16930 width=N) (actual rows=16930.00 loops=1) +(1 row) + DROP FUNCTION explain_mask_costs(text, bool, bool, bool, bool); diff --git a/src/test/regress/sql/planner_est.sql b/src/test/regress/sql/planner_est.sql index 2b696a4e4e5..e5e42ad8d0d 100644 --- a/src/test/regress/sql/planner_est.sql +++ b/src/test/regress/sql/planner_est.sql @@ -153,4 +153,29 @@ CREATE TEMP TABLE char_table_1 AS ANALYZE char_table_1; EXPLAIN (COSTS OFF) SELECT * FROM char_table_1 WHERE c < 'Q'; +-- +-- Test IS [NOT] DISTINCT FROM join estimates with NULLs on both sides. Only +-- the top plan line is shown, since the rest is not stable across platforms. +-- +CREATE TEMP TABLE distinct_t1 AS + SELECT CASE WHEN i > 30 THEN i END AS a FROM generate_series(1, 100) i; +CREATE TEMP TABLE distinct_t2 AS + SELECT CASE WHEN i > 100 THEN i - 100 END AS a FROM generate_series(1, 200) i; +ANALYZE distinct_t1, distinct_t2; + +-- Ensure pairs of NULLs are counted +SELECT * FROM explain_mask_costs($$ +SELECT * FROM distinct_t1 t1 JOIN distinct_t2 t2 ON t1.a IS NOT DISTINCT FROM t2.a;$$, +true, true, false, true) LIMIT 1; + +-- The equivalent OR clause should get about the same estimate +SELECT * FROM explain_mask_costs($$ +SELECT * FROM distinct_t1 t1 JOIN distinct_t2 t2 ON t1.a = t2.a OR (t1.a IS NULL AND t2.a IS NULL);$$, +true, true, false, true) LIMIT 1; + +-- Ensure pairs of NULLs are not counted +SELECT * FROM explain_mask_costs($$ +SELECT * FROM distinct_t1 t1 JOIN distinct_t2 t2 ON t1.a IS DISTINCT FROM t2.a;$$, +true, true, false, true) LIMIT 1; + DROP FUNCTION explain_mask_costs(text, bool, bool, bool, bool); -- 2.37.1 (Apple Git-137.1)