From 7f6d7f1a4be893a840e9f11820ef9ef9c4c8ccc4 Mon Sep 17 00:00:00 2001 From: shihao zhong Date: Mon, 5 Oct 2026 11:41:09 -0400 Subject: [PATCH v1 2/2] Add a test for Memoize with mismatched collations This is a LEFT JOIN from a case insensitive column to a case sensitive one, with an explicit COLLATE in the join clause. Without the fix, 'abc' and 'ABC' share one cache entry and the counts come out wrong. --- .../regress/expected/collate.icu.utf8.out | 40 +++++++++++++++++++ src/test/regress/sql/collate.icu.utf8.sql | 22 ++++++++++ 2 files changed, 62 insertions(+) diff --git a/src/test/regress/expected/collate.icu.utf8.out b/src/test/regress/expected/collate.icu.utf8.out index 5ec848b82e4..fa45c4803ff 100644 --- a/src/test/regress/expected/collate.icu.utf8.out +++ b/src/test/regress/expected/collate.icu.utf8.out @@ -2195,6 +2195,46 @@ WHERE x COLLATE case_insensitive IN (SELECT x FROM test3cs); 3 (1 row) +ROLLBACK; +-- Memoize must not share a cache entry between 'abc' and 'ABC' when the join +-- clause can tell them apart. It is only chosen when the outer side has +-- statistics, hence the new table. +BEGIN; +CREATE TABLE test_memoize_ci (x text COLLATE case_insensitive); +INSERT INTO test_memoize_ci SELECT x FROM test3ci, generate_series(1, 10); +ANALYZE test_memoize_ci; +SET LOCAL enable_hashjoin TO off; +SET LOCAL enable_mergejoin TO off; +EXPLAIN (COSTS OFF) +SELECT t2.x, count(*) FROM test_memoize_ci t1 + LEFT JOIN test3cs t2 ON t2.x COLLATE case_sensitive = t1.x +GROUP BY t2.x ORDER BY t2.x; + QUERY PLAN +--------------------------------------------------------------------------- + Sort + Sort Key: t2.x COLLATE case_sensitive + -> HashAggregate + Group Key: t2.x + -> Nested Loop Left Join + -> Seq Scan on test_memoize_ci t1 + -> Memoize + Cache Key: t1.x + Cache Mode: logical + -> Index Only Scan using test3cs_x_idx on test3cs t2 + Index Cond: (x = t1.x) +(11 rows) + +SELECT t2.x, count(*) FROM test_memoize_ci t1 + LEFT JOIN test3cs t2 ON t2.x COLLATE case_sensitive = t1.x +GROUP BY t2.x ORDER BY t2.x; + x | count +-----+------- + abc | 10 + ABC | 10 + def | 10 + ghi | 10 +(4 rows) + ROLLBACK; -- These queries should be able to use the index on test1ci.x: SET enable_seqscan = off; diff --git a/src/test/regress/sql/collate.icu.utf8.sql b/src/test/regress/sql/collate.icu.utf8.sql index b50cf3f6c7c..601ebdff1fb 100644 --- a/src/test/regress/sql/collate.icu.utf8.sql +++ b/src/test/regress/sql/collate.icu.utf8.sql @@ -796,6 +796,28 @@ WHERE x COLLATE case_insensitive IN (SELECT x FROM test3cs); ROLLBACK; +-- Memoize must not share a cache entry between 'abc' and 'ABC' when the join +-- clause can tell them apart. It is only chosen when the outer side has +-- statistics, hence the new table. +BEGIN; + +CREATE TABLE test_memoize_ci (x text COLLATE case_insensitive); +INSERT INTO test_memoize_ci SELECT x FROM test3ci, generate_series(1, 10); +ANALYZE test_memoize_ci; + +SET LOCAL enable_hashjoin TO off; +SET LOCAL enable_mergejoin TO off; + +EXPLAIN (COSTS OFF) +SELECT t2.x, count(*) FROM test_memoize_ci t1 + LEFT JOIN test3cs t2 ON t2.x COLLATE case_sensitive = t1.x +GROUP BY t2.x ORDER BY t2.x; +SELECT t2.x, count(*) FROM test_memoize_ci t1 + LEFT JOIN test3cs t2 ON t2.x COLLATE case_sensitive = t1.x +GROUP BY t2.x ORDER BY t2.x; + +ROLLBACK; + -- These queries should be able to use the index on test1ci.x: SET enable_seqscan = off; SET enable_indexonlyscan = off; -- 2.37.1 (Apple Git-137.1)