From 3be0cc13fb47810538bfcc7f6bee51eb0f746891 Mon Sep 17 00:00:00 2001
From: Jelte Fennema-Nio <postgres@jeltef.nl>
Date: Sat, 26 Sep 2026 15:37:08 +0200
Subject: [PATCH v2 4/4] postgres_fdw: Fix costing of remote sorts without
 remote estimates

Without use_remote_estimate, postgres_fdw has no way to know what a remote
sort costs. The heuristic used so far was to multiply the path's cost by
DEFAULT_FDW_SORT_MULTIPLIER (1.2). Commit f18c944b61 introduced it and
its message explains that the intent of that constant was to prefer a
remote sort over a local one (if a sort is useful). In practice that
doesn't actually work in lots of cases though.

The surcharge has no relation to the number of rows being sorted, so for an
expensive path that produces few rows, such as an aggregate or a join, it is
arbitrarily larger than the cost of actually sorting the output. Since the
alternative, a local Sort over the unsorted foreign path, is costed
accurately, the pushed-down sort always lost in those cases. See the
expected regress output changes in the patch for examples.

The later commit ffab494a4d ran into this issue too[1]. It tried to fix
this in two ways depending on the situation:

1. By calculating what a local sort would be and using that same value
   for the remote.
2. By reducing DEFAULT_FDW_SORT_MULTIPLIER to 1.05 in one place in the
   code, which was noted in the thread as being chosen fairly
   arbitrarily to improve some plans[1].

This commit generalizes that first approach and uses it for every sorted
foreign path, with one slight improvement: Instead of using the full cost of
the local sort, it's multiplied by a fraction (0.8). That way the remote
and local sort don't tie, but the remote sort is preferred. This answers
the open question from [1]: no percentage of the path cost is reasonable,
because the surcharge should scale with the sort, not with the path.

This changes a bunch of plans in our existing tests for the better:

1. Pushing down a Sort node to the remote side
2. Changing a local merge join on top of remotely sorted scan to a hash
   join over unsorted remote scan.
3. Changing a local merge append over multiple remotely sorted scans to
   a hash aggregate over unsorted remote scans.

[1]: https://postgr.es/m/5C232F39.9060509@lab.ntt.co.jp

Discussion: https://postgr.es/m/DLRIUVXIEUOQ.2E2I0B7VZLHLI@jeltef.nl
---
 .../postgres_fdw/expected/postgres_fdw.out    | 264 +++++++++---------
 contrib/postgres_fdw/postgres_fdw.c           | 151 ++++------
 contrib/postgres_fdw/sql/postgres_fdw.sql     |   6 +-
 3 files changed, 193 insertions(+), 228 deletions(-)

diff --git a/contrib/postgres_fdw/expected/postgres_fdw.out b/contrib/postgres_fdw/expected/postgres_fdw.out
index 5df5cbdf472..c56f3b9da30 100644
--- a/contrib/postgres_fdw/expected/postgres_fdw.out
+++ b/contrib/postgres_fdw/expected/postgres_fdw.out
@@ -2037,21 +2037,18 @@ SELECT t1.c1, t2.c2, t3.c3 FROM ft2 t1 LEFT JOIN ft2 t2 ON (t1.c1 = t2.c1) RIGHT
  40 |  0 | AAA040
 (10 rows)
 
--- full outer join + WHERE clause, only matched rows
+-- full outer join + WHERE clause, only matched rows.  The ORDER BY and LIMIT
+-- are pushed down too: without remote estimates, a remote sort should be
+-- preferred over a local one.
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1, t2.c1 FROM ft4 t1 FULL JOIN ft5 t2 ON (t1.c1 = t2.c1) WHERE (t1.c1 = t2.c1 OR t1.c1 IS NULL) ORDER BY t1.c1, t2.c1 OFFSET 10 LIMIT 10;
-                                                                            QUERY PLAN                                                                            
-------------------------------------------------------------------------------------------------------------------------------------------------------------------
- Limit
+                                                                                                                 QUERY PLAN                                                                                                                  
+---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: t1.c1, t2.c1
-   ->  Sort
-         Output: t1.c1, t2.c1
-         Sort Key: t1.c1, t2.c1
-         ->  Foreign Scan
-               Output: t1.c1, t2.c1
-               Relations: (public.ft4 t1) FULL JOIN (public.ft5 t2)
-               Remote SQL: SELECT r1.c1, r2.c1 FROM ("S 1"."T 3" r1 FULL JOIN "S 1"."T 4" r2 ON (((r1.c1 = r2.c1)))) WHERE (((r1.c1 = r2.c1) OR (r1.c1 IS NULL)))
-(9 rows)
+   Relations: (public.ft4 t1) FULL JOIN (public.ft5 t2)
+   Remote SQL: SELECT r1.c1, r2.c1 FROM ("S 1"."T 3" r1 FULL JOIN "S 1"."T 4" r2 ON (((r1.c1 = r2.c1)))) WHERE (((r1.c1 = r2.c1) OR (r1.c1 IS NULL))) ORDER BY r1.c1 ASC NULLS LAST, r2.c1 ASC NULLS LAST LIMIT 10::bigint OFFSET 10::bigint
+(4 rows)
 
 SELECT t1.c1, t2.c1 FROM ft4 t1 FULL JOIN ft5 t2 ON (t1.c1 = t2.c1) WHERE (t1.c1 = t2.c1 OR t1.c1 IS NULL) ORDER BY t1.c1, t2.c1 OFFSET 10 LIMIT 10;
  c1 | c1 
@@ -3011,16 +3008,13 @@ ALTER VIEW v4 OWNER TO regress_view_owner;
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1, t1.c3 FROM ft1 t1, unnest(ARRAY[1, 5, 10, 100]::int[]) AS u(id)
 WHERE t1.c1 = u.id ORDER BY t1.c1;
-                                                                   QUERY PLAN                                                                   
-------------------------------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                                QUERY PLAN                                                                                 
+---------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: t1.c1, t1.c3
-   Sort Key: t1.c1
-   ->  Foreign Scan
-         Output: t1.c1, t1.c3
-         Relations: (public.ft1 t1) INNER JOIN (pg_catalog.unnest() u)
-         Remote SQL: SELECT r1."C 1", r1.c3 FROM ("S 1"."T 1" r1 INNER JOIN unnest('{1,5,10,100}'::integer[]) f2(c1) ON (((r1."C 1" = f2.c1))))
-(7 rows)
+   Relations: (public.ft1 t1) INNER JOIN (pg_catalog.unnest() u)
+   Remote SQL: SELECT r1."C 1", r1.c3 FROM ("S 1"."T 1" r1 INNER JOIN unnest('{1,5,10,100}'::integer[]) f2(c1) ON (((r1."C 1" = f2.c1)))) ORDER BY r1."C 1" ASC NULLS LAST
+(4 rows)
 
 SELECT t1.c1, t1.c3 FROM ft1 t1, unnest(ARRAY[1, 5, 10, 100]::int[]) AS u(id)
 WHERE t1.c1 = u.id ORDER BY t1.c1;
@@ -3036,16 +3030,13 @@ WHERE t1.c1 = u.id ORDER BY t1.c1;
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1 FROM ft1 t1, generate_series(1, 4) AS g(id)
 WHERE t1.c1 = g.id ORDER BY t1.c1;
-                                                         QUERY PLAN                                                          
------------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                       QUERY PLAN                                                                       
+--------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: t1.c1
-   Sort Key: t1.c1
-   ->  Foreign Scan
-         Output: t1.c1
-         Relations: (public.ft1 t1) INNER JOIN (pg_catalog.generate_series() g)
-         Remote SQL: SELECT r1."C 1" FROM ("S 1"."T 1" r1 INNER JOIN generate_series(1, 4) f2(c1) ON (((r1."C 1" = f2.c1))))
-(7 rows)
+   Relations: (public.ft1 t1) INNER JOIN (pg_catalog.generate_series() g)
+   Remote SQL: SELECT r1."C 1" FROM ("S 1"."T 1" r1 INNER JOIN generate_series(1, 4) f2(c1) ON (((r1."C 1" = f2.c1)))) ORDER BY r1."C 1" ASC NULLS LAST
+(4 rows)
 
 SELECT t1.c1 FROM ft1 t1, generate_series(1, 4) AS g(id)
 WHERE t1.c1 = g.id ORDER BY t1.c1;
@@ -3129,20 +3120,22 @@ FROM ft1 t1, ft6 t2, unnest(ARRAY[1, 5, 10, 100]::int[]) AS u(id)
 WHERE t1.c1 = u.id AND t2.c1 = u.id ORDER BY t1.c1;
                                                                    QUERY PLAN                                                                   
 ------------------------------------------------------------------------------------------------------------------------------------------------
- Merge Join
+ Sort
    Output: t1.c1, t2.c1
-   Merge Cond: (t1.c1 = u.id)
-   ->  Foreign Scan on public.ft1 t1
-         Output: t1.c1
-         Remote SQL: SELECT "C 1" FROM "S 1"."T 1" ORDER BY "C 1" ASC NULLS LAST
-   ->  Sort
-         Output: t2.c1, u.id
-         Sort Key: t2.c1
+   Sort Key: t1.c1
+   ->  Hash Join
+         Output: t1.c1, t2.c1
+         Hash Cond: (u.id = t1.c1)
          ->  Foreign Scan
                Output: t2.c1, u.id
                Relations: (public.ft6 t2) INNER JOIN (pg_catalog.unnest() u)
                Remote SQL: SELECT r2.c1, f3.c1 FROM ("S 1"."T 4" r2 INNER JOIN unnest('{1,5,10,100}'::integer[]) f3(c1) ON (((r2.c1 = f3.c1))))
-(13 rows)
+         ->  Hash
+               Output: t1.c1
+               ->  Foreign Scan on public.ft1 t1
+                     Output: t1.c1
+                     Remote SQL: SELECT "C 1" FROM "S 1"."T 1"
+(15 rows)
 
 -- Cost-based selection between two foreign servers: ft1 ("S 1"."T 1") has
 -- 1000 rows, ft6 ("S 1"."T 4") has ~33 rows.  The same query shape gets a
@@ -3384,16 +3377,13 @@ SELECT r.a, t.n, t.s
   FROM remote_tbl r, ROWS FROM (unnest(array[3, 6, 9]),
                                 generate_series(11, 13)) AS t(n, s)
  WHERE r.a = t.n ORDER BY r.a;
-                                                                                        QUERY PLAN                                                                                        
-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                                                   QUERY PLAN                                                                                                    
+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: r.a, t.n, t.s
-   Sort Key: r.a
-   ->  Foreign Scan
-         Output: r.a, t.n, t.s
-         Relations: (public.remote_tbl r) INNER JOIN (ROWS FROM (pg_catalog.unnest(), pg_catalog.generate_series()) t)
-         Remote SQL: SELECT r1.a, f2.c1, f2.c2 FROM (public.base_tbl_fn r1 INNER JOIN ROWS FROM (unnest('{3,6,9}'::integer[]), generate_series(11, 13)) f2(c1, c2) ON (((r1.a = f2.c1))))
-(7 rows)
+   Relations: (public.remote_tbl r) INNER JOIN (ROWS FROM (pg_catalog.unnest(), pg_catalog.generate_series()) t)
+   Remote SQL: SELECT r1.a, f2.c1, f2.c2 FROM (public.base_tbl_fn r1 INNER JOIN ROWS FROM (unnest('{3,6,9}'::integer[]), generate_series(11, 13)) f2(c1, c2) ON (((r1.a = f2.c1)))) ORDER BY r1.a ASC NULLS LAST
+(4 rows)
 
 SELECT r.a, t.n, t.s
   FROM remote_tbl r, ROWS FROM (unnest(array[3, 6, 9]),
@@ -3414,16 +3404,13 @@ SELECT t::text, r.a
   FROM remote_tbl r, ROWS FROM (unnest(array[3, 6, 9]),
                                 generate_series(11, 13)) AS t(n, s)
  WHERE r.a = t.n ORDER BY r.a;
-                                                                                                                QUERY PLAN                                                                                                                 
--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
- Sort
-   Output: ((t.*)::text), r.a
-   Sort Key: r.a
-   ->  Foreign Scan
-         Output: (t.*)::text, r.a
-         Relations: (public.remote_tbl r) INNER JOIN (ROWS FROM (pg_catalog.unnest(), pg_catalog.generate_series()) t)
-         Remote SQL: SELECT CASE WHEN (f2.*)::text IS NOT NULL THEN ROW(f2.c1, f2.c2) END, r1.a FROM (public.base_tbl_fn r1 INNER JOIN ROWS FROM (unnest('{3,6,9}'::integer[]), generate_series(11, 13)) f2(c1, c2) ON (((r1.a = f2.c1))))
-(7 rows)
+                                                                                                                            QUERY PLAN                                                                                                                            
+------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
+   Output: (t.*)::text, r.a
+   Relations: (public.remote_tbl r) INNER JOIN (ROWS FROM (pg_catalog.unnest(), pg_catalog.generate_series()) t)
+   Remote SQL: SELECT CASE WHEN (f2.*)::text IS NOT NULL THEN ROW(f2.c1, f2.c2) END, r1.a FROM (public.base_tbl_fn r1 INNER JOIN ROWS FROM (unnest('{3,6,9}'::integer[]), generate_series(11, 13)) f2(c1, c2) ON (((r1.a = f2.c1)))) ORDER BY r1.a ASC NULLS LAST
+(4 rows)
 
 SELECT t::text, r.a
   FROM remote_tbl r, ROWS FROM (unnest(array[3, 6, 9]),
@@ -3611,21 +3598,23 @@ EXPLAIN (VERBOSE, COSTS OFF)
 SELECT r.a FROM remote_tbl r
  WHERE EXISTS (SELECT 1 FROM unnest(array[3, 6, 9]) AS t(n) WHERE t.n = r.a)
  ORDER BY r.a;
-                                   QUERY PLAN                                   
---------------------------------------------------------------------------------
- Merge Semi Join
+                           QUERY PLAN                            
+-----------------------------------------------------------------
+ Sort
    Output: r.a
-   Merge Cond: (r.a = t.n)
-   ->  Foreign Scan on public.remote_tbl r
-         Output: r.a, r.b
-         Remote SQL: SELECT a FROM public.base_tbl_fn ORDER BY a ASC NULLS LAST
-   ->  Sort
-         Output: t.n
-         Sort Key: t.n
-         ->  Function Scan on pg_catalog.unnest t
+   Sort Key: r.a
+   ->  Hash Semi Join
+         Output: r.a
+         Hash Cond: (r.a = t.n)
+         ->  Foreign Scan on public.remote_tbl r
+               Output: r.a, r.b
+               Remote SQL: SELECT a FROM public.base_tbl_fn
+         ->  Hash
                Output: t.n
-               Function Call: unnest('{3,6,9}'::integer[])
-(12 rows)
+               ->  Function Scan on pg_catalog.unnest t
+                     Output: t.n
+                     Function Call: unnest('{3,6,9}'::integer[])
+(14 rows)
 
 DROP FOREIGN TABLE remote_tbl;
 DROP TABLE base_tbl_fn;
@@ -4211,16 +4200,13 @@ select sum(c2) filter (where c2 in (select c2 from ft1 where c2 < 5)) from ft1;
 -- Ordered-sets within aggregate
 explain (verbose, costs off)
 select c2, rank('10'::varchar) within group (order by c6), percentile_cont(c2/10::numeric) within group (order by c1) from ft1 where c2 < 10 group by c2 having percentile_cont(c2/10::numeric) within group (order by c1) < 500 order by c2;
-                                                                                                                                                                           QUERY PLAN                                                                                                                                                                           
-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                                                                                                                                     QUERY PLAN                                                                                                                                                                                      
+-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: c2, (rank('10'::character varying) WITHIN GROUP (ORDER BY c6)), (percentile_cont((((c2)::numeric / '10'::numeric))::double precision) WITHIN GROUP (ORDER BY ((c1)::double precision)))
-   Sort Key: ft1.c2
-   ->  Foreign Scan
-         Output: c2, (rank('10'::character varying) WITHIN GROUP (ORDER BY c6)), (percentile_cont((((c2)::numeric / '10'::numeric))::double precision) WITHIN GROUP (ORDER BY ((c1)::double precision)))
-         Relations: Aggregate on (public.ft1)
-         Remote SQL: SELECT c2, rank('10'::character varying) WITHIN GROUP (ORDER BY c6 ASC NULLS LAST), percentile_cont((c2 / 10::numeric)) WITHIN GROUP (ORDER BY ("C 1") ASC NULLS LAST) FROM "S 1"."T 1" WHERE ((c2 < 10)) GROUP BY 1 HAVING ((percentile_cont((c2 / 10::numeric)) WITHIN GROUP (ORDER BY ("C 1") ASC NULLS LAST) < 500::double precision))
-(7 rows)
+   Relations: Aggregate on (public.ft1)
+   Remote SQL: SELECT c2, rank('10'::character varying) WITHIN GROUP (ORDER BY c6 ASC NULLS LAST), percentile_cont((c2 / 10::numeric)) WITHIN GROUP (ORDER BY ("C 1") ASC NULLS LAST) FROM "S 1"."T 1" WHERE ((c2 < 10)) GROUP BY 1 HAVING ((percentile_cont((c2 / 10::numeric)) WITHIN GROUP (ORDER BY ("C 1") ASC NULLS LAST) < 500::double precision)) ORDER BY c2 ASC NULLS LAST
+(4 rows)
 
 select c2, rank('10'::varchar) within group (order by c6), percentile_cont(c2/10::numeric) within group (order by c1) from ft1 where c2 < 10 group by c2 having percentile_cont(c2/10::numeric) within group (order by c1) < 500 order by c2;
  c2 | rank | percentile_cont 
@@ -4276,18 +4262,17 @@ alter extension postgres_fdw add function least_accum(anyelement, variadic anyar
 alter extension postgres_fdw add aggregate least_agg(variadic items anyarray);
 alter server loopback options (set extensions 'postgres_fdw');
 -- Now aggregate will be pushed.  Aggregate will display VARIADIC argument.
+-- The ORDER BY is pushed down along with it: without remote estimates, a
+-- remote sort should be preferred over a local one.
 explain (verbose, costs off)
 select c2, least_agg(c1) from ft1 where c2 < 100 group by c2 order by c2;
-                                                      QUERY PLAN                                                       
------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                 QUERY PLAN                                                                 
+--------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: c2, (least_agg(VARIADIC ARRAY[c1]))
-   Sort Key: ft1.c2
-   ->  Foreign Scan
-         Output: c2, (least_agg(VARIADIC ARRAY[c1]))
-         Relations: Aggregate on (public.ft1)
-         Remote SQL: SELECT c2, public.least_agg(VARIADIC ARRAY["C 1"]) FROM "S 1"."T 1" WHERE ((c2 < 100)) GROUP BY 1
-(7 rows)
+   Relations: Aggregate on (public.ft1)
+   Remote SQL: SELECT c2, public.least_agg(VARIADIC ARRAY["C 1"]) FROM "S 1"."T 1" WHERE ((c2 < 100)) GROUP BY 1 ORDER BY c2 ASC NULLS LAST
+(4 rows)
 
 select c2, least_agg(c1) from ft1 where c2 < 100 group by c2 order by c2;
  c2 | least_agg 
@@ -11372,16 +11357,18 @@ ANALYZE fpagg_tab_p3;
 SET enable_partitionwise_aggregate TO false;
 EXPLAIN (COSTS OFF)
 SELECT a, sum(b), min(b), count(*) FROM pagg_tab GROUP BY a HAVING avg(b) < 22 ORDER BY 1;
-                     QUERY PLAN                      
------------------------------------------------------
- GroupAggregate
-   Group Key: pagg_tab.a
-   Filter: (avg(pagg_tab.b) < '22'::numeric)
-   ->  Append
-         ->  Foreign Scan on fpagg_tab_p1 pagg_tab_1
-         ->  Foreign Scan on fpagg_tab_p2 pagg_tab_2
-         ->  Foreign Scan on fpagg_tab_p3 pagg_tab_3
-(7 rows)
+                        QUERY PLAN                         
+-----------------------------------------------------------
+ Sort
+   Sort Key: pagg_tab.a
+   ->  HashAggregate
+         Group Key: pagg_tab.a
+         Filter: (avg(pagg_tab.b) < '22'::numeric)
+         ->  Append
+               ->  Foreign Scan on fpagg_tab_p1 pagg_tab_1
+               ->  Foreign Scan on fpagg_tab_p2 pagg_tab_2
+               ->  Foreign Scan on fpagg_tab_p3 pagg_tab_3
+(9 rows)
 
 -- Plan with partitionwise aggregates is enabled
 SET enable_partitionwise_aggregate TO true;
@@ -11415,32 +11402,34 @@ SELECT a, sum(b), min(b), count(*) FROM pagg_tab GROUP BY a HAVING avg(b) < 22 O
 -- Should have all the columns in the target list for the given relation
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT a, count(t1) FROM pagg_tab t1 GROUP BY a HAVING avg(b) < 22 ORDER BY 1;
-                                         QUERY PLAN                                         
---------------------------------------------------------------------------------------------
- Merge Append
+                               QUERY PLAN                               
+------------------------------------------------------------------------
+ Sort
+   Output: t1.a, (count(((t1.*)::pagg_tab)))
    Sort Key: t1.a
-   ->  GroupAggregate
-         Output: t1.a, count(((t1.*)::pagg_tab))
-         Group Key: t1.a
-         Filter: (avg(t1.b) < '22'::numeric)
-         ->  Foreign Scan on public.fpagg_tab_p1 t1
-               Output: t1.a, t1.*, t1.b
-               Remote SQL: SELECT a, b, c FROM public.pagg_tab_p1 ORDER BY a ASC NULLS LAST
-   ->  GroupAggregate
-         Output: t1_1.a, count(((t1_1.*)::pagg_tab))
-         Group Key: t1_1.a
-         Filter: (avg(t1_1.b) < '22'::numeric)
-         ->  Foreign Scan on public.fpagg_tab_p2 t1_1
-               Output: t1_1.a, t1_1.*, t1_1.b
-               Remote SQL: SELECT a, b, c FROM public.pagg_tab_p2 ORDER BY a ASC NULLS LAST
-   ->  GroupAggregate
-         Output: t1_2.a, count(((t1_2.*)::pagg_tab))
-         Group Key: t1_2.a
-         Filter: (avg(t1_2.b) < '22'::numeric)
-         ->  Foreign Scan on public.fpagg_tab_p3 t1_2
-               Output: t1_2.a, t1_2.*, t1_2.b
-               Remote SQL: SELECT a, b, c FROM public.pagg_tab_p3 ORDER BY a ASC NULLS LAST
-(23 rows)
+   ->  Append
+         ->  HashAggregate
+               Output: t1.a, count(((t1.*)::pagg_tab))
+               Group Key: t1.a
+               Filter: (avg(t1.b) < '22'::numeric)
+               ->  Foreign Scan on public.fpagg_tab_p1 t1
+                     Output: t1.a, t1.*, t1.b
+                     Remote SQL: SELECT a, b, c FROM public.pagg_tab_p1
+         ->  HashAggregate
+               Output: t1_1.a, count(((t1_1.*)::pagg_tab))
+               Group Key: t1_1.a
+               Filter: (avg(t1_1.b) < '22'::numeric)
+               ->  Foreign Scan on public.fpagg_tab_p2 t1_1
+                     Output: t1_1.a, t1_1.*, t1_1.b
+                     Remote SQL: SELECT a, b, c FROM public.pagg_tab_p2
+         ->  HashAggregate
+               Output: t1_2.a, count(((t1_2.*)::pagg_tab))
+               Group Key: t1_2.a
+               Filter: (avg(t1_2.b) < '22'::numeric)
+               ->  Foreign Scan on public.fpagg_tab_p3 t1_2
+                     Output: t1_2.a, t1_2.*, t1_2.b
+                     Remote SQL: SELECT a, b, c FROM public.pagg_tab_p3
+(25 rows)
 
 SELECT a, count(t1) FROM pagg_tab t1 GROUP BY a HAVING avg(b) < 22 ORDER BY 1;
  a  | count 
@@ -11456,23 +11445,24 @@ SELECT a, count(t1) FROM pagg_tab t1 GROUP BY a HAVING avg(b) < 22 ORDER BY 1;
 -- When GROUP BY clause does not match with PARTITION KEY.
 EXPLAIN (COSTS OFF)
 SELECT b, avg(a), max(a), count(*) FROM pagg_tab GROUP BY b HAVING sum(a) < 700 ORDER BY 1;
-                        QUERY PLAN                         
------------------------------------------------------------
+                           QUERY PLAN                            
+-----------------------------------------------------------------
  Finalize GroupAggregate
    Group Key: pagg_tab.b
    Filter: (sum(pagg_tab.a) < 700)
-   ->  Merge Append
+   ->  Sort
          Sort Key: pagg_tab.b
-         ->  Partial GroupAggregate
-               Group Key: pagg_tab.b
-               ->  Foreign Scan on fpagg_tab_p1 pagg_tab
-         ->  Partial GroupAggregate
-               Group Key: pagg_tab_1.b
-               ->  Foreign Scan on fpagg_tab_p2 pagg_tab_1
-         ->  Partial GroupAggregate
-               Group Key: pagg_tab_2.b
-               ->  Foreign Scan on fpagg_tab_p3 pagg_tab_2
-(14 rows)
+         ->  Append
+               ->  Partial HashAggregate
+                     Group Key: pagg_tab.b
+                     ->  Foreign Scan on fpagg_tab_p1 pagg_tab
+               ->  Partial HashAggregate
+                     Group Key: pagg_tab_1.b
+                     ->  Foreign Scan on fpagg_tab_p2 pagg_tab_1
+               ->  Partial HashAggregate
+                     Group Key: pagg_tab_2.b
+                     ->  Foreign Scan on fpagg_tab_p3 pagg_tab_2
+(15 rows)
 
 -- ===================================================================
 -- access rights and superuser
diff --git a/contrib/postgres_fdw/postgres_fdw.c b/contrib/postgres_fdw/postgres_fdw.c
index 101d161cb68..6d6ab64ce92 100644
--- a/contrib/postgres_fdw/postgres_fdw.c
+++ b/contrib/postgres_fdw/postgres_fdw.c
@@ -68,8 +68,15 @@ PG_MODULE_MAGIC_EXT(
 /* Default CPU cost to process 1 row (above and beyond cpu_tuple_cost). */
 #define DEFAULT_FDW_TUPLE_COST		0.2
 
-/* If no remote estimates, assume a sort costs 20% extra */
-#define DEFAULT_FDW_SORT_MULTIPLIER 1.2
+/*
+ * If no remote estimates, charge this fraction of a local sort's cost for a
+ * remote sort.  It needs to be clearly below 1 so that a remote sort beats a
+ * local Sort of the same rows, and clearly above 0 so that a sorted path
+ * doesn't look free when the ordering is only potentially useful (e.g. for a
+ * merge join).  Within that range the exact value is arbitrary.  See
+ * adjust_foreign_path_cost_for_sort().
+ */
+#define DEFAULT_FDW_SORT_COST_FRACTION 0.8
 
 /*
  * Indexes of FDW-private information stored in fdw_private lists.
@@ -524,12 +531,9 @@ static void get_remote_estimate(const char *sql,
 								int *width,
 								Cost *startup_cost,
 								Cost *total_cost);
-static void adjust_foreign_grouping_path_cost(PlannerInfo *root,
-											  List *pathkeys,
-											  double retrieved_rows,
-											  double width,
+static void adjust_foreign_path_cost_for_sort(PlannerInfo *root, List *pathkeys,
+											  double retrieved_rows, double width,
 											  double limit_tuples,
-											  int *p_disabled_nodes,
 											  Cost *p_startup_cost,
 											  Cost *p_run_cost);
 static bool ec_member_matches_foreign(PlannerInfo *root, RelOptInfo *rel,
@@ -3754,38 +3758,14 @@ estimate_path_cost_size(PlannerInfo *root,
 
 		/*
 		 * Without remote estimates, we have no real way to estimate the cost
-		 * of generating sorted output.  It could be free if the query plan
-		 * the remote side would have chosen generates properly-sorted output
-		 * anyway, but in most cases it will cost something.  Estimate a value
-		 * high enough that we won't pick the sorted path when the ordering
-		 * isn't locally useful, but low enough that we'll err on the side of
-		 * pushing down the ORDER BY clause when it's useful to do so.
+		 * of generating sorted output, so use a heuristic.  See
+		 * adjust_foreign_path_cost_for_sort().
 		 */
 		if (pathkeys != NIL)
-		{
-			if (IS_UPPER_REL(foreignrel))
-			{
-				Assert(foreignrel->reloptkind == RELOPT_UPPER_REL &&
-					   fpinfo->stage == UPPERREL_GROUP_AGG);
-
-				/*
-				 * We can only get here when this function is called from
-				 * add_foreign_ordered_paths() or add_foreign_final_paths();
-				 * in which cases, the passed-in fpextra should not be NULL.
-				 */
-				Assert(fpextra);
-				adjust_foreign_grouping_path_cost(root, pathkeys,
-												  retrieved_rows, width,
-												  fpextra->limit_tuples,
-												  &disabled_nodes,
-												  &startup_cost, &run_cost);
-			}
-			else
-			{
-				startup_cost *= DEFAULT_FDW_SORT_MULTIPLIER;
-				run_cost *= DEFAULT_FDW_SORT_MULTIPLIER;
-			}
-		}
+			adjust_foreign_path_cost_for_sort(root, pathkeys,
+											  retrieved_rows, width,
+											  fpextra ? fpextra->limit_tuples : -1.0,
+											  &startup_cost, &run_cost);
 
 		/* Adjust the cost estimates if we have LIMIT */
 		if (fpextra && fpextra->has_limit)
@@ -3874,11 +3854,10 @@ estimate_path_cost_size(PlannerInfo *root,
 	 * restriction (see create_limit_path()), there would be no difference
 	 * between the costs of the local restriction and the costs of the remote
 	 * restriction estimated above if we don't use remote estimates (except
-	 * for the case where the foreignrel is a grouping relation, the given
-	 * pathkeys is not NIL, and the effects of a bounded sort for that rel is
-	 * accounted for in costing the remote restriction).  Tweak the costs of
-	 * the remote restriction to ensure we'll prefer it if LIMIT is a useful
-	 * one.
+	 * for the case where the given pathkeys is not NIL, and the effects of a
+	 * bounded sort are accounted for in costing the remote restriction).
+	 * Tweak the costs of the remote restriction to ensure we'll prefer it if
+	 * LIMIT is a useful one.
 	 */
 	if (!fpinfo->use_remote_estimate &&
 		fpextra && fpextra->has_limit &&
@@ -3936,58 +3915,50 @@ get_remote_estimate(const char *sql, PGconn *conn,
 }
 
 /*
- * Adjust the cost estimates of a foreign grouping path to include the cost of
- * generating properly-sorted output.
+ * Adjust the given path costs for having the remote side sort its output,
+ * when no remote estimates are available.
+ *
+ * We can't accurately estimate a remote sort; it might even be free if the
+ * remote plan happens to produce the order (e.g. a sorted aggregate).  What we
+ * do know is that the alternative is the unsorted path plus a local Sort of
+ * the same rows. We assume that the remote can sort at least as cheaply as we
+ * can.  So charge a fraction of the cost of a local Sort: enough to beat it,
+ * not so little that the sorted path looks free.
+ *
+ * Like a local Sort, a remote sort is blocking, so all of the input cost
+ * becomes startup cost.  Besides being accurate, that also matters when the
+ * ordering is merely potentially useful (e.g. for a merge join): the sorted
+ * path then shares a pathlist with the unsorted path, and add_path() lets
+ * fuzzily equal costs be decided by pathkeys.  A surcharge on the run cost
+ * alone can be within that fuzz for a small table, so the sorted path would
+ * prune the unsorted one and force every consumer to sort remotely.  With
+ * the input cost moved to startup, the sorted path is always clearly worse
+ * on startup cost, so both survive and the consumer gets to choose.
  */
 static void
-adjust_foreign_grouping_path_cost(PlannerInfo *root,
-								  List *pathkeys,
-								  double retrieved_rows,
-								  double width,
+adjust_foreign_path_cost_for_sort(PlannerInfo *root, List *pathkeys,
+								  double retrieved_rows, double width,
 								  double limit_tuples,
-								  int *p_disabled_nodes,
-								  Cost *p_startup_cost,
-								  Cost *p_run_cost)
+								  Cost *p_startup_cost, Cost *p_run_cost)
 {
-	/*
-	 * If the GROUP BY clause isn't sort-able, the plan chosen by the remote
-	 * side is unlikely to generate properly-sorted output, so it would need
-	 * an explicit sort; adjust the given costs with cost_sort().  Likewise,
-	 * if the GROUP BY clause is sort-able but isn't a superset of the given
-	 * pathkeys, adjust the costs with that function.  Otherwise, adjust the
-	 * costs by applying the same heuristic as for the scan or join case.
-	 */
-	if (!grouping_is_sortable(root->processed_groupClause) ||
-		!pathkeys_contained_in(pathkeys, root->group_pathkeys))
-	{
-		Path		sort_path;	/* dummy for result of cost_sort */
-
-		cost_sort(&sort_path,
-				  root,
-				  pathkeys,
-				  0,
-				  *p_startup_cost + *p_run_cost,
-				  retrieved_rows,
-				  width,
-				  0.0,
-				  work_mem,
-				  limit_tuples);
-
-		*p_startup_cost = sort_path.startup_cost;
-		*p_run_cost = sort_path.total_cost - sort_path.startup_cost;
-	}
-	else
-	{
-		/*
-		 * The default extra cost seems too large for foreign-grouping cases;
-		 * add 1/4th of that default.
-		 */
-		double		sort_multiplier = 1.0 + (DEFAULT_FDW_SORT_MULTIPLIER
-											 - 1.0) * 0.25;
-
-		*p_startup_cost *= sort_multiplier;
-		*p_run_cost *= sort_multiplier;
-	}
+	Cost		input_cost = *p_startup_cost + *p_run_cost;
+	Path		sort_path;		/* dummy for result of cost_sort */
+
+	cost_sort(&sort_path,
+			  root,
+			  pathkeys,
+			  0,
+			  input_cost,
+			  retrieved_rows,
+			  width,
+			  0.0,
+			  work_mem,
+			  limit_tuples);
+
+	*p_startup_cost = input_cost +
+		(sort_path.startup_cost - input_cost) * DEFAULT_FDW_SORT_COST_FRACTION;
+	*p_run_cost = (sort_path.total_cost - sort_path.startup_cost) *
+		DEFAULT_FDW_SORT_COST_FRACTION;
 }
 
 /*
diff --git a/contrib/postgres_fdw/sql/postgres_fdw.sql b/contrib/postgres_fdw/sql/postgres_fdw.sql
index 8e9186be92b..0745e36135c 100644
--- a/contrib/postgres_fdw/sql/postgres_fdw.sql
+++ b/contrib/postgres_fdw/sql/postgres_fdw.sql
@@ -681,7 +681,9 @@ RESET enable_memoize;
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1, t2.c2, t3.c3 FROM ft2 t1 LEFT JOIN ft2 t2 ON (t1.c1 = t2.c1) RIGHT JOIN ft4 t3 ON (t2.c1 = t3.c1) OFFSET 10 LIMIT 10;
 SELECT t1.c1, t2.c2, t3.c3 FROM ft2 t1 LEFT JOIN ft2 t2 ON (t1.c1 = t2.c1) RIGHT JOIN ft4 t3 ON (t2.c1 = t3.c1) OFFSET 10 LIMIT 10;
--- full outer join + WHERE clause, only matched rows
+-- full outer join + WHERE clause, only matched rows.  The ORDER BY and LIMIT
+-- are pushed down too: without remote estimates, a remote sort should be
+-- preferred over a local one.
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1, t2.c1 FROM ft4 t1 FULL JOIN ft5 t2 ON (t1.c1 = t2.c1) WHERE (t1.c1 = t2.c1 OR t1.c1 IS NULL) ORDER BY t1.c1, t2.c1 OFFSET 10 LIMIT 10;
 SELECT t1.c1, t2.c1 FROM ft4 t1 FULL JOIN ft5 t2 ON (t1.c1 = t2.c1) WHERE (t1.c1 = t2.c1 OR t1.c1 IS NULL) ORDER BY t1.c1, t2.c1 OFFSET 10 LIMIT 10;
@@ -1304,6 +1306,8 @@ alter extension postgres_fdw add aggregate least_agg(variadic items anyarray);
 alter server loopback options (set extensions 'postgres_fdw');
 
 -- Now aggregate will be pushed.  Aggregate will display VARIADIC argument.
+-- The ORDER BY is pushed down along with it: without remote estimates, a
+-- remote sort should be preferred over a local one.
 explain (verbose, costs off)
 select c2, least_agg(c1) from ft1 where c2 < 100 group by c2 order by c2;
 select c2, least_agg(c1) from ft1 where c2 < 100 group by c2 order by c2;
-- 
2.55.0

