From eec35b8190820ea41db9778f7c7820ac8a292aec Mon Sep 17 00:00:00 2001
From: Jelte Fennema-Nio <postgres@jeltef.nl>
Date: Sat, 26 Sep 2026 15:37:08 +0200
Subject: [PATCH v1] postgres_fdw: Fix costing of remote sorts without remote
 estimates

Without use_remote_estimate, postgres_fdw has no way to know what a remote
sort costs. The heuristic used so far was to multiply the path's cost by
DEFAULT_FDW_SORT_MULTIPLIER (1.2). Commit f18c944b61 introduced it and
its message explains that the intent of that constant was to prefer a
remote sort over a local one (if a sort is useful). In practice that
doesn't actually work in lots of cases though.

The surcharge has no relation to the number of rows being sorted, so for an
expensive path that produces few rows, such as an aggregate or a join, it is
arbitrarily larger than the cost of actually sorting the output. Since the
alternative, a local Sort over the unsorted foreign path, is costed
accurately, the pushed-down sort always lost in those cases. See the
expected regress output changes in the patch for examples.

The later commit ffab494a4d ran into this issue too[1]. It tried to fix
this in two ways depending on the situation:

1. By calculating what a local sort would be and using that same value
   for the remote.
2. By reducing DEFAULT_FDW_SORT_MULTIPLIER to 1.05 in one place in the
   code, which was noted in the thread as being chosen fairly
   arbitrarily to improve some plans[1].

This commit generalizes that first approach and uses it for every sorted
foreign path, with one slight improvement: Instead of using the full cost of
the local sort, it's multiplied by a fraction (0.8). That way the remote
and local sort don't tie, but the remote sort is preferred. This answers
the open question from [1]: no percentage of the path cost is reasonable,
because the surcharge should scale with the sort, not with the path.

This changes a bunch of plans in our existing tests for the better:

1. Pushing down a Sort node to the remote side
2. Changing a local merge join on top of remotely sorted scan to a hash
   join over unsorted remote scan.
3. Changing a local merge append over multiple remotely sorted scans to
   a hash aggregate over unsorted remote scans.

[1]: https://postgr.es/m/5C232F39.9060509@lab.ntt.co.jp
---
 .../postgres_fdw/expected/postgres_fdw.out    | 264 +++++++++---------
 contrib/postgres_fdw/postgres_fdw.c           | 131 ++++-----
 contrib/postgres_fdw/sql/postgres_fdw.sql     |   6 +-
 3 files changed, 184 insertions(+), 217 deletions(-)

diff --git a/contrib/postgres_fdw/expected/postgres_fdw.out b/contrib/postgres_fdw/expected/postgres_fdw.out
index 739f43af7bb..e087bf8260c 100644
--- a/contrib/postgres_fdw/expected/postgres_fdw.out
+++ b/contrib/postgres_fdw/expected/postgres_fdw.out
@@ -1971,21 +1971,18 @@ SELECT t1.c1, t2.c2, t3.c3 FROM ft2 t1 LEFT JOIN ft2 t2 ON (t1.c1 = t2.c1) RIGHT
  40 |  0 | AAA040
 (10 rows)
 
--- full outer join + WHERE clause, only matched rows
+-- full outer join + WHERE clause, only matched rows.  The ORDER BY and LIMIT
+-- are pushed down too: without remote estimates, a remote sort should be
+-- preferred over a local one.
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1, t2.c1 FROM ft4 t1 FULL JOIN ft5 t2 ON (t1.c1 = t2.c1) WHERE (t1.c1 = t2.c1 OR t1.c1 IS NULL) ORDER BY t1.c1, t2.c1 OFFSET 10 LIMIT 10;
-                                                                            QUERY PLAN                                                                            
-------------------------------------------------------------------------------------------------------------------------------------------------------------------
- Limit
+                                                                                                                 QUERY PLAN                                                                                                                  
+---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: t1.c1, t2.c1
-   ->  Sort
-         Output: t1.c1, t2.c1
-         Sort Key: t1.c1, t2.c1
-         ->  Foreign Scan
-               Output: t1.c1, t2.c1
-               Relations: (public.ft4 t1) FULL JOIN (public.ft5 t2)
-               Remote SQL: SELECT r1.c1, r2.c1 FROM ("S 1"."T 3" r1 FULL JOIN "S 1"."T 4" r2 ON (((r1.c1 = r2.c1)))) WHERE (((r1.c1 = r2.c1) OR (r1.c1 IS NULL)))
-(9 rows)
+   Relations: (public.ft4 t1) FULL JOIN (public.ft5 t2)
+   Remote SQL: SELECT r1.c1, r2.c1 FROM ("S 1"."T 3" r1 FULL JOIN "S 1"."T 4" r2 ON (((r1.c1 = r2.c1)))) WHERE (((r1.c1 = r2.c1) OR (r1.c1 IS NULL))) ORDER BY r1.c1 ASC NULLS LAST, r2.c1 ASC NULLS LAST LIMIT 10::bigint OFFSET 10::bigint
+(4 rows)
 
 SELECT t1.c1, t2.c1 FROM ft4 t1 FULL JOIN ft5 t2 ON (t1.c1 = t2.c1) WHERE (t1.c1 = t2.c1 OR t1.c1 IS NULL) ORDER BY t1.c1, t2.c1 OFFSET 10 LIMIT 10;
  c1 | c1 
@@ -2945,16 +2942,13 @@ ALTER VIEW v4 OWNER TO regress_view_owner;
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1, t1.c3 FROM ft1 t1, unnest(ARRAY[1, 5, 10, 100]::int[]) AS u(id)
 WHERE t1.c1 = u.id ORDER BY t1.c1;
-                                                                   QUERY PLAN                                                                   
-------------------------------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                                QUERY PLAN                                                                                 
+---------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: t1.c1, t1.c3
-   Sort Key: t1.c1
-   ->  Foreign Scan
-         Output: t1.c1, t1.c3
-         Relations: (public.ft1 t1) INNER JOIN (pg_catalog.unnest() u)
-         Remote SQL: SELECT r1."C 1", r1.c3 FROM ("S 1"."T 1" r1 INNER JOIN unnest('{1,5,10,100}'::integer[]) f2(c1) ON (((r1."C 1" = f2.c1))))
-(7 rows)
+   Relations: (public.ft1 t1) INNER JOIN (pg_catalog.unnest() u)
+   Remote SQL: SELECT r1."C 1", r1.c3 FROM ("S 1"."T 1" r1 INNER JOIN unnest('{1,5,10,100}'::integer[]) f2(c1) ON (((r1."C 1" = f2.c1)))) ORDER BY r1."C 1" ASC NULLS LAST
+(4 rows)
 
 SELECT t1.c1, t1.c3 FROM ft1 t1, unnest(ARRAY[1, 5, 10, 100]::int[]) AS u(id)
 WHERE t1.c1 = u.id ORDER BY t1.c1;
@@ -2970,16 +2964,13 @@ WHERE t1.c1 = u.id ORDER BY t1.c1;
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1 FROM ft1 t1, generate_series(1, 4) AS g(id)
 WHERE t1.c1 = g.id ORDER BY t1.c1;
-                                                         QUERY PLAN                                                          
------------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                       QUERY PLAN                                                                       
+--------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: t1.c1
-   Sort Key: t1.c1
-   ->  Foreign Scan
-         Output: t1.c1
-         Relations: (public.ft1 t1) INNER JOIN (pg_catalog.generate_series() g)
-         Remote SQL: SELECT r1."C 1" FROM ("S 1"."T 1" r1 INNER JOIN generate_series(1, 4) f2(c1) ON (((r1."C 1" = f2.c1))))
-(7 rows)
+   Relations: (public.ft1 t1) INNER JOIN (pg_catalog.generate_series() g)
+   Remote SQL: SELECT r1."C 1" FROM ("S 1"."T 1" r1 INNER JOIN generate_series(1, 4) f2(c1) ON (((r1."C 1" = f2.c1)))) ORDER BY r1."C 1" ASC NULLS LAST
+(4 rows)
 
 SELECT t1.c1 FROM ft1 t1, generate_series(1, 4) AS g(id)
 WHERE t1.c1 = g.id ORDER BY t1.c1;
@@ -3063,20 +3054,22 @@ FROM ft1 t1, ft6 t2, unnest(ARRAY[1, 5, 10, 100]::int[]) AS u(id)
 WHERE t1.c1 = u.id AND t2.c1 = u.id ORDER BY t1.c1;
                                                                    QUERY PLAN                                                                   
 ------------------------------------------------------------------------------------------------------------------------------------------------
- Merge Join
+ Sort
    Output: t1.c1, t2.c1
-   Merge Cond: (t1.c1 = u.id)
-   ->  Foreign Scan on public.ft1 t1
-         Output: t1.c1
-         Remote SQL: SELECT "C 1" FROM "S 1"."T 1" ORDER BY "C 1" ASC NULLS LAST
-   ->  Sort
-         Output: t2.c1, u.id
-         Sort Key: t2.c1
+   Sort Key: t1.c1
+   ->  Hash Join
+         Output: t1.c1, t2.c1
+         Hash Cond: (u.id = t1.c1)
          ->  Foreign Scan
                Output: t2.c1, u.id
                Relations: (public.ft6 t2) INNER JOIN (pg_catalog.unnest() u)
                Remote SQL: SELECT r2.c1, f3.c1 FROM ("S 1"."T 4" r2 INNER JOIN unnest('{1,5,10,100}'::integer[]) f3(c1) ON (((r2.c1 = f3.c1))))
-(13 rows)
+         ->  Hash
+               Output: t1.c1
+               ->  Foreign Scan on public.ft1 t1
+                     Output: t1.c1
+                     Remote SQL: SELECT "C 1" FROM "S 1"."T 1"
+(15 rows)
 
 -- Cost-based selection between two foreign servers: ft1 ("S 1"."T 1") has
 -- 1000 rows, ft6 ("S 1"."T 4") has ~33 rows.  The same query shape gets a
@@ -3318,16 +3311,13 @@ SELECT r.a, t.n, t.s
   FROM remote_tbl r, ROWS FROM (unnest(array[3, 6, 9]),
                                 generate_series(11, 13)) AS t(n, s)
  WHERE r.a = t.n ORDER BY r.a;
-                                                                                        QUERY PLAN                                                                                        
-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                                                   QUERY PLAN                                                                                                    
+-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: r.a, t.n, t.s
-   Sort Key: r.a
-   ->  Foreign Scan
-         Output: r.a, t.n, t.s
-         Relations: (public.remote_tbl r) INNER JOIN (ROWS FROM (pg_catalog.unnest(), pg_catalog.generate_series()) t)
-         Remote SQL: SELECT r1.a, f2.c1, f2.c2 FROM (public.base_tbl_fn r1 INNER JOIN ROWS FROM (unnest('{3,6,9}'::integer[]), generate_series(11, 13)) f2(c1, c2) ON (((r1.a = f2.c1))))
-(7 rows)
+   Relations: (public.remote_tbl r) INNER JOIN (ROWS FROM (pg_catalog.unnest(), pg_catalog.generate_series()) t)
+   Remote SQL: SELECT r1.a, f2.c1, f2.c2 FROM (public.base_tbl_fn r1 INNER JOIN ROWS FROM (unnest('{3,6,9}'::integer[]), generate_series(11, 13)) f2(c1, c2) ON (((r1.a = f2.c1)))) ORDER BY r1.a ASC NULLS LAST
+(4 rows)
 
 SELECT r.a, t.n, t.s
   FROM remote_tbl r, ROWS FROM (unnest(array[3, 6, 9]),
@@ -3348,16 +3338,13 @@ SELECT t::text, r.a
   FROM remote_tbl r, ROWS FROM (unnest(array[3, 6, 9]),
                                 generate_series(11, 13)) AS t(n, s)
  WHERE r.a = t.n ORDER BY r.a;
-                                                                                                                QUERY PLAN                                                                                                                 
--------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
- Sort
-   Output: ((t.*)::text), r.a
-   Sort Key: r.a
-   ->  Foreign Scan
-         Output: (t.*)::text, r.a
-         Relations: (public.remote_tbl r) INNER JOIN (ROWS FROM (pg_catalog.unnest(), pg_catalog.generate_series()) t)
-         Remote SQL: SELECT CASE WHEN (f2.*)::text IS NOT NULL THEN ROW(f2.c1, f2.c2) END, r1.a FROM (public.base_tbl_fn r1 INNER JOIN ROWS FROM (unnest('{3,6,9}'::integer[]), generate_series(11, 13)) f2(c1, c2) ON (((r1.a = f2.c1))))
-(7 rows)
+                                                                                                                            QUERY PLAN                                                                                                                            
+------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
+   Output: (t.*)::text, r.a
+   Relations: (public.remote_tbl r) INNER JOIN (ROWS FROM (pg_catalog.unnest(), pg_catalog.generate_series()) t)
+   Remote SQL: SELECT CASE WHEN (f2.*)::text IS NOT NULL THEN ROW(f2.c1, f2.c2) END, r1.a FROM (public.base_tbl_fn r1 INNER JOIN ROWS FROM (unnest('{3,6,9}'::integer[]), generate_series(11, 13)) f2(c1, c2) ON (((r1.a = f2.c1)))) ORDER BY r1.a ASC NULLS LAST
+(4 rows)
 
 SELECT t::text, r.a
   FROM remote_tbl r, ROWS FROM (unnest(array[3, 6, 9]),
@@ -3545,21 +3532,23 @@ EXPLAIN (VERBOSE, COSTS OFF)
 SELECT r.a FROM remote_tbl r
  WHERE EXISTS (SELECT 1 FROM unnest(array[3, 6, 9]) AS t(n) WHERE t.n = r.a)
  ORDER BY r.a;
-                                   QUERY PLAN                                   
---------------------------------------------------------------------------------
- Merge Semi Join
+                           QUERY PLAN                            
+-----------------------------------------------------------------
+ Sort
    Output: r.a
-   Merge Cond: (r.a = t.n)
-   ->  Foreign Scan on public.remote_tbl r
-         Output: r.a, r.b
-         Remote SQL: SELECT a FROM public.base_tbl_fn ORDER BY a ASC NULLS LAST
-   ->  Sort
-         Output: t.n
-         Sort Key: t.n
-         ->  Function Scan on pg_catalog.unnest t
+   Sort Key: r.a
+   ->  Hash Semi Join
+         Output: r.a
+         Hash Cond: (r.a = t.n)
+         ->  Foreign Scan on public.remote_tbl r
+               Output: r.a, r.b
+               Remote SQL: SELECT a FROM public.base_tbl_fn
+         ->  Hash
                Output: t.n
-               Function Call: unnest('{3,6,9}'::integer[])
-(12 rows)
+               ->  Function Scan on pg_catalog.unnest t
+                     Output: t.n
+                     Function Call: unnest('{3,6,9}'::integer[])
+(14 rows)
 
 DROP FOREIGN TABLE remote_tbl;
 DROP TABLE base_tbl_fn;
@@ -4145,16 +4134,13 @@ select sum(c2) filter (where c2 in (select c2 from ft1 where c2 < 5)) from ft1;
 -- Ordered-sets within aggregate
 explain (verbose, costs off)
 select c2, rank('10'::varchar) within group (order by c6), percentile_cont(c2/10::numeric) within group (order by c1) from ft1 where c2 < 10 group by c2 having percentile_cont(c2/10::numeric) within group (order by c1) < 500 order by c2;
-                                                                                                                                                                           QUERY PLAN                                                                                                                                                                           
-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                                                                                                                                     QUERY PLAN                                                                                                                                                                                      
+-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: c2, (rank('10'::character varying) WITHIN GROUP (ORDER BY c6)), (percentile_cont((((c2)::numeric / '10'::numeric))::double precision) WITHIN GROUP (ORDER BY ((c1)::double precision)))
-   Sort Key: ft1.c2
-   ->  Foreign Scan
-         Output: c2, (rank('10'::character varying) WITHIN GROUP (ORDER BY c6)), (percentile_cont((((c2)::numeric / '10'::numeric))::double precision) WITHIN GROUP (ORDER BY ((c1)::double precision)))
-         Relations: Aggregate on (public.ft1)
-         Remote SQL: SELECT c2, rank('10'::character varying) WITHIN GROUP (ORDER BY c6 ASC NULLS LAST), percentile_cont((c2 / 10::numeric)) WITHIN GROUP (ORDER BY ("C 1") ASC NULLS LAST) FROM "S 1"."T 1" WHERE ((c2 < 10)) GROUP BY 1 HAVING ((percentile_cont((c2 / 10::numeric)) WITHIN GROUP (ORDER BY ("C 1") ASC NULLS LAST) < 500::double precision))
-(7 rows)
+   Relations: Aggregate on (public.ft1)
+   Remote SQL: SELECT c2, rank('10'::character varying) WITHIN GROUP (ORDER BY c6 ASC NULLS LAST), percentile_cont((c2 / 10::numeric)) WITHIN GROUP (ORDER BY ("C 1") ASC NULLS LAST) FROM "S 1"."T 1" WHERE ((c2 < 10)) GROUP BY 1 HAVING ((percentile_cont((c2 / 10::numeric)) WITHIN GROUP (ORDER BY ("C 1") ASC NULLS LAST) < 500::double precision)) ORDER BY c2 ASC NULLS LAST
+(4 rows)
 
 select c2, rank('10'::varchar) within group (order by c6), percentile_cont(c2/10::numeric) within group (order by c1) from ft1 where c2 < 10 group by c2 having percentile_cont(c2/10::numeric) within group (order by c1) < 500 order by c2;
  c2 | rank | percentile_cont 
@@ -4210,18 +4196,17 @@ alter extension postgres_fdw add function least_accum(anyelement, variadic anyar
 alter extension postgres_fdw add aggregate least_agg(variadic items anyarray);
 alter server loopback options (set extensions 'postgres_fdw');
 -- Now aggregate will be pushed.  Aggregate will display VARIADIC argument.
+-- The ORDER BY is pushed down along with it: without remote estimates, a
+-- remote sort should be preferred over a local one.
 explain (verbose, costs off)
 select c2, least_agg(c1) from ft1 where c2 < 100 group by c2 order by c2;
-                                                      QUERY PLAN                                                       
------------------------------------------------------------------------------------------------------------------------
- Sort
+                                                                 QUERY PLAN                                                                 
+--------------------------------------------------------------------------------------------------------------------------------------------
+ Foreign Scan
    Output: c2, (least_agg(VARIADIC ARRAY[c1]))
-   Sort Key: ft1.c2
-   ->  Foreign Scan
-         Output: c2, (least_agg(VARIADIC ARRAY[c1]))
-         Relations: Aggregate on (public.ft1)
-         Remote SQL: SELECT c2, public.least_agg(VARIADIC ARRAY["C 1"]) FROM "S 1"."T 1" WHERE ((c2 < 100)) GROUP BY 1
-(7 rows)
+   Relations: Aggregate on (public.ft1)
+   Remote SQL: SELECT c2, public.least_agg(VARIADIC ARRAY["C 1"]) FROM "S 1"."T 1" WHERE ((c2 < 100)) GROUP BY 1 ORDER BY c2 ASC NULLS LAST
+(4 rows)
 
 select c2, least_agg(c1) from ft1 where c2 < 100 group by c2 order by c2;
  c2 | least_agg 
@@ -11306,16 +11291,18 @@ ANALYZE fpagg_tab_p3;
 SET enable_partitionwise_aggregate TO false;
 EXPLAIN (COSTS OFF)
 SELECT a, sum(b), min(b), count(*) FROM pagg_tab GROUP BY a HAVING avg(b) < 22 ORDER BY 1;
-                     QUERY PLAN                      
------------------------------------------------------
- GroupAggregate
-   Group Key: pagg_tab.a
-   Filter: (avg(pagg_tab.b) < '22'::numeric)
-   ->  Append
-         ->  Foreign Scan on fpagg_tab_p1 pagg_tab_1
-         ->  Foreign Scan on fpagg_tab_p2 pagg_tab_2
-         ->  Foreign Scan on fpagg_tab_p3 pagg_tab_3
-(7 rows)
+                        QUERY PLAN                         
+-----------------------------------------------------------
+ Sort
+   Sort Key: pagg_tab.a
+   ->  HashAggregate
+         Group Key: pagg_tab.a
+         Filter: (avg(pagg_tab.b) < '22'::numeric)
+         ->  Append
+               ->  Foreign Scan on fpagg_tab_p1 pagg_tab_1
+               ->  Foreign Scan on fpagg_tab_p2 pagg_tab_2
+               ->  Foreign Scan on fpagg_tab_p3 pagg_tab_3
+(9 rows)
 
 -- Plan with partitionwise aggregates is enabled
 SET enable_partitionwise_aggregate TO true;
@@ -11349,32 +11336,34 @@ SELECT a, sum(b), min(b), count(*) FROM pagg_tab GROUP BY a HAVING avg(b) < 22 O
 -- Should have all the columns in the target list for the given relation
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT a, count(t1) FROM pagg_tab t1 GROUP BY a HAVING avg(b) < 22 ORDER BY 1;
-                                         QUERY PLAN                                         
---------------------------------------------------------------------------------------------
- Merge Append
+                               QUERY PLAN                               
+------------------------------------------------------------------------
+ Sort
+   Output: t1.a, (count(((t1.*)::pagg_tab)))
    Sort Key: t1.a
-   ->  GroupAggregate
-         Output: t1.a, count(((t1.*)::pagg_tab))
-         Group Key: t1.a
-         Filter: (avg(t1.b) < '22'::numeric)
-         ->  Foreign Scan on public.fpagg_tab_p1 t1
-               Output: t1.a, t1.*, t1.b
-               Remote SQL: SELECT a, b, c FROM public.pagg_tab_p1 ORDER BY a ASC NULLS LAST
-   ->  GroupAggregate
-         Output: t1_1.a, count(((t1_1.*)::pagg_tab))
-         Group Key: t1_1.a
-         Filter: (avg(t1_1.b) < '22'::numeric)
-         ->  Foreign Scan on public.fpagg_tab_p2 t1_1
-               Output: t1_1.a, t1_1.*, t1_1.b
-               Remote SQL: SELECT a, b, c FROM public.pagg_tab_p2 ORDER BY a ASC NULLS LAST
-   ->  GroupAggregate
-         Output: t1_2.a, count(((t1_2.*)::pagg_tab))
-         Group Key: t1_2.a
-         Filter: (avg(t1_2.b) < '22'::numeric)
-         ->  Foreign Scan on public.fpagg_tab_p3 t1_2
-               Output: t1_2.a, t1_2.*, t1_2.b
-               Remote SQL: SELECT a, b, c FROM public.pagg_tab_p3 ORDER BY a ASC NULLS LAST
-(23 rows)
+   ->  Append
+         ->  HashAggregate
+               Output: t1.a, count(((t1.*)::pagg_tab))
+               Group Key: t1.a
+               Filter: (avg(t1.b) < '22'::numeric)
+               ->  Foreign Scan on public.fpagg_tab_p1 t1
+                     Output: t1.a, t1.*, t1.b
+                     Remote SQL: SELECT a, b, c FROM public.pagg_tab_p1
+         ->  HashAggregate
+               Output: t1_1.a, count(((t1_1.*)::pagg_tab))
+               Group Key: t1_1.a
+               Filter: (avg(t1_1.b) < '22'::numeric)
+               ->  Foreign Scan on public.fpagg_tab_p2 t1_1
+                     Output: t1_1.a, t1_1.*, t1_1.b
+                     Remote SQL: SELECT a, b, c FROM public.pagg_tab_p2
+         ->  HashAggregate
+               Output: t1_2.a, count(((t1_2.*)::pagg_tab))
+               Group Key: t1_2.a
+               Filter: (avg(t1_2.b) < '22'::numeric)
+               ->  Foreign Scan on public.fpagg_tab_p3 t1_2
+                     Output: t1_2.a, t1_2.*, t1_2.b
+                     Remote SQL: SELECT a, b, c FROM public.pagg_tab_p3
+(25 rows)
 
 SELECT a, count(t1) FROM pagg_tab t1 GROUP BY a HAVING avg(b) < 22 ORDER BY 1;
  a  | count 
@@ -11390,23 +11379,24 @@ SELECT a, count(t1) FROM pagg_tab t1 GROUP BY a HAVING avg(b) < 22 ORDER BY 1;
 -- When GROUP BY clause does not match with PARTITION KEY.
 EXPLAIN (COSTS OFF)
 SELECT b, avg(a), max(a), count(*) FROM pagg_tab GROUP BY b HAVING sum(a) < 700 ORDER BY 1;
-                        QUERY PLAN                         
------------------------------------------------------------
+                           QUERY PLAN                            
+-----------------------------------------------------------------
  Finalize GroupAggregate
    Group Key: pagg_tab.b
    Filter: (sum(pagg_tab.a) < 700)
-   ->  Merge Append
+   ->  Sort
          Sort Key: pagg_tab.b
-         ->  Partial GroupAggregate
-               Group Key: pagg_tab.b
-               ->  Foreign Scan on fpagg_tab_p1 pagg_tab
-         ->  Partial GroupAggregate
-               Group Key: pagg_tab_1.b
-               ->  Foreign Scan on fpagg_tab_p2 pagg_tab_1
-         ->  Partial GroupAggregate
-               Group Key: pagg_tab_2.b
-               ->  Foreign Scan on fpagg_tab_p3 pagg_tab_2
-(14 rows)
+         ->  Append
+               ->  Partial HashAggregate
+                     Group Key: pagg_tab.b
+                     ->  Foreign Scan on fpagg_tab_p1 pagg_tab
+               ->  Partial HashAggregate
+                     Group Key: pagg_tab_1.b
+                     ->  Foreign Scan on fpagg_tab_p2 pagg_tab_1
+               ->  Partial HashAggregate
+                     Group Key: pagg_tab_2.b
+                     ->  Foreign Scan on fpagg_tab_p3 pagg_tab_2
+(15 rows)
 
 -- ===================================================================
 -- access rights and superuser
diff --git a/contrib/postgres_fdw/postgres_fdw.c b/contrib/postgres_fdw/postgres_fdw.c
index 2bcff4b26b4..40f93fb6960 100644
--- a/contrib/postgres_fdw/postgres_fdw.c
+++ b/contrib/postgres_fdw/postgres_fdw.c
@@ -68,8 +68,11 @@ PG_MODULE_MAGIC_EXT(
 /* Default CPU cost to process 1 row (above and beyond cpu_tuple_cost). */
 #define DEFAULT_FDW_TUPLE_COST		0.2
 
-/* If no remote estimates, assume a sort costs 20% extra */
-#define DEFAULT_FDW_SORT_MULTIPLIER 1.2
+/*
+ * If no remote estimates, charge this fraction of a local sort's cost for a
+ * remote sort.  See adjust_foreign_path_cost_for_sort().
+ */
+#define DEFAULT_FDW_SORT_COST_FRACTION 0.8
 
 /*
  * Indexes of FDW-private information stored in fdw_private lists.
@@ -524,12 +527,9 @@ static void get_remote_estimate(const char *sql,
 								int *width,
 								Cost *startup_cost,
 								Cost *total_cost);
-static void adjust_foreign_grouping_path_cost(PlannerInfo *root,
-											  List *pathkeys,
-											  double retrieved_rows,
-											  double width,
+static void adjust_foreign_path_cost_for_sort(PlannerInfo *root, List *pathkeys,
+											  double retrieved_rows, double width,
 											  double limit_tuples,
-											  int *p_disabled_nodes,
 											  Cost *p_startup_cost,
 											  Cost *p_run_cost);
 static bool ec_member_matches_foreign(PlannerInfo *root, RelOptInfo *rel,
@@ -3802,30 +3802,10 @@ estimate_path_cost_size(PlannerInfo *root,
 		 * pushing down the ORDER BY clause when it's useful to do so.
 		 */
 		if (pathkeys != NIL)
-		{
-			if (IS_UPPER_REL(foreignrel))
-			{
-				Assert(foreignrel->reloptkind == RELOPT_UPPER_REL &&
-					   fpinfo->stage == UPPERREL_GROUP_AGG);
-
-				/*
-				 * We can only get here when this function is called from
-				 * add_foreign_ordered_paths() or add_foreign_final_paths();
-				 * in which cases, the passed-in fpextra should not be NULL.
-				 */
-				Assert(fpextra);
-				adjust_foreign_grouping_path_cost(root, pathkeys,
-												  retrieved_rows, width,
-												  fpextra->limit_tuples,
-												  &disabled_nodes,
-												  &startup_cost, &run_cost);
-			}
-			else
-			{
-				startup_cost *= DEFAULT_FDW_SORT_MULTIPLIER;
-				run_cost *= DEFAULT_FDW_SORT_MULTIPLIER;
-			}
-		}
+			adjust_foreign_path_cost_for_sort(root, pathkeys,
+											  retrieved_rows, width,
+											  fpextra ? fpextra->limit_tuples : -1.0,
+											  &startup_cost, &run_cost);
 
 		total_cost = startup_cost + run_cost;
 
@@ -3955,58 +3935,51 @@ get_remote_estimate(const char *sql, PGconn *conn,
 }
 
 /*
- * Adjust the cost estimates of a foreign grouping path to include the cost of
- * generating properly-sorted output.
+ * Adjust the given path costs for having the remote side sort its output,
+ * when no remote estimates are available.
+ *
+ * We can't accurately estimate a remote sort; it might even be free if the
+ * remote plan happens to produce the order (e.g. a sorted aggregate).  What we
+ * do know is that the alternative is the unsorted path plus a local Sort of
+ * the same rows, which is costed accurately. We assume that the remote can
+ * sort at least as cheaply as we can.  So charge a fraction of the cost of a
+ * local Sort: enough to beat it, not so little that the sorted path looks
+ * free.
+ *
+ * Like a local Sort, a remote sort is blocking, so all of the input cost
+ * becomes startup cost.  Besides being accurate, that also matters when the
+ * ordering is merely potentially useful (e.g. for a merge join): the sorted
+ * path then shares a pathlist with the unsorted path, and add_path() lets
+ * fuzzily equal costs be decided by pathkeys.  A surcharge on the run cost
+ * alone can be within that fuzz for a small table, so the sorted path would
+ * prune the unsorted one and force every consumer to sort remotely.  With
+ * the input cost moved to startup, the sorted path is always clearly worse
+ * on startup cost, so both survive and the consumer gets to choose.
  */
 static void
-adjust_foreign_grouping_path_cost(PlannerInfo *root,
-								  List *pathkeys,
-								  double retrieved_rows,
-								  double width,
+adjust_foreign_path_cost_for_sort(PlannerInfo *root, List *pathkeys,
+								  double retrieved_rows, double width,
 								  double limit_tuples,
-								  int *p_disabled_nodes,
-								  Cost *p_startup_cost,
-								  Cost *p_run_cost)
+								  Cost *p_startup_cost, Cost *p_run_cost)
 {
-	/*
-	 * If the GROUP BY clause isn't sort-able, the plan chosen by the remote
-	 * side is unlikely to generate properly-sorted output, so it would need
-	 * an explicit sort; adjust the given costs with cost_sort().  Likewise,
-	 * if the GROUP BY clause is sort-able but isn't a superset of the given
-	 * pathkeys, adjust the costs with that function.  Otherwise, adjust the
-	 * costs by applying the same heuristic as for the scan or join case.
-	 */
-	if (!grouping_is_sortable(root->processed_groupClause) ||
-		!pathkeys_contained_in(pathkeys, root->group_pathkeys))
-	{
-		Path		sort_path;	/* dummy for result of cost_sort */
-
-		cost_sort(&sort_path,
-				  root,
-				  pathkeys,
-				  0,
-				  *p_startup_cost + *p_run_cost,
-				  retrieved_rows,
-				  width,
-				  0.0,
-				  work_mem,
-				  limit_tuples);
-
-		*p_startup_cost = sort_path.startup_cost;
-		*p_run_cost = sort_path.total_cost - sort_path.startup_cost;
-	}
-	else
-	{
-		/*
-		 * The default extra cost seems too large for foreign-grouping cases;
-		 * add 1/4th of that default.
-		 */
-		double		sort_multiplier = 1.0 + (DEFAULT_FDW_SORT_MULTIPLIER
-											 - 1.0) * 0.25;
-
-		*p_startup_cost *= sort_multiplier;
-		*p_run_cost *= sort_multiplier;
-	}
+	Cost		input_cost = *p_startup_cost + *p_run_cost;
+	Path		sort_path;		/* dummy for result of cost_sort */
+
+	cost_sort(&sort_path,
+			  root,
+			  pathkeys,
+			  0,
+			  input_cost,
+			  retrieved_rows,
+			  width,
+			  0.0,
+			  work_mem,
+			  limit_tuples);
+
+	*p_startup_cost = input_cost +
+		(sort_path.startup_cost - input_cost) * DEFAULT_FDW_SORT_COST_FRACTION;
+	*p_run_cost = (sort_path.total_cost - sort_path.startup_cost) *
+		DEFAULT_FDW_SORT_COST_FRACTION;
 }
 
 /*
diff --git a/contrib/postgres_fdw/sql/postgres_fdw.sql b/contrib/postgres_fdw/sql/postgres_fdw.sql
index f1ca3204382..e858f740d6e 100644
--- a/contrib/postgres_fdw/sql/postgres_fdw.sql
+++ b/contrib/postgres_fdw/sql/postgres_fdw.sql
@@ -646,7 +646,9 @@ RESET enable_memoize;
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1, t2.c2, t3.c3 FROM ft2 t1 LEFT JOIN ft2 t2 ON (t1.c1 = t2.c1) RIGHT JOIN ft4 t3 ON (t2.c1 = t3.c1) OFFSET 10 LIMIT 10;
 SELECT t1.c1, t2.c2, t3.c3 FROM ft2 t1 LEFT JOIN ft2 t2 ON (t1.c1 = t2.c1) RIGHT JOIN ft4 t3 ON (t2.c1 = t3.c1) OFFSET 10 LIMIT 10;
--- full outer join + WHERE clause, only matched rows
+-- full outer join + WHERE clause, only matched rows.  The ORDER BY and LIMIT
+-- are pushed down too: without remote estimates, a remote sort should be
+-- preferred over a local one.
 EXPLAIN (VERBOSE, COSTS OFF)
 SELECT t1.c1, t2.c1 FROM ft4 t1 FULL JOIN ft5 t2 ON (t1.c1 = t2.c1) WHERE (t1.c1 = t2.c1 OR t1.c1 IS NULL) ORDER BY t1.c1, t2.c1 OFFSET 10 LIMIT 10;
 SELECT t1.c1, t2.c1 FROM ft4 t1 FULL JOIN ft5 t2 ON (t1.c1 = t2.c1) WHERE (t1.c1 = t2.c1 OR t1.c1 IS NULL) ORDER BY t1.c1, t2.c1 OFFSET 10 LIMIT 10;
@@ -1269,6 +1271,8 @@ alter extension postgres_fdw add aggregate least_agg(variadic items anyarray);
 alter server loopback options (set extensions 'postgres_fdw');
 
 -- Now aggregate will be pushed.  Aggregate will display VARIADIC argument.
+-- The ORDER BY is pushed down along with it: without remote estimates, a
+-- remote sort should be preferred over a local one.
 explain (verbose, costs off)
 select c2, least_agg(c1) from ft1 where c2 < 100 group by c2 order by c2;
 select c2, least_agg(c1) from ft1 where c2 < 100 group by c2 order by c2;
-- 
2.55.0

