From a1f4a758f811b0697a0ba4ab4263b2a8edafe499 Mon Sep 17 00:00:00 2001
From: Greg Burd <greg@burd.me>
Date: Wed, 7 Oct 2026 01:50:09 +0000
Subject: [PATCH v5 3/4] Let an ordering index scan below a join emit its ORDER
 BY values

With the previous two commits, an ordering Index Scan returns its ORDER
BY values to a target list entry that matches the ORDER BY expression,
and isn't charged for it.  But that only happens when the scan is the
top of the scan/join tree.  When the scan is below a join, its target
is rel->reltarget, which holds only the Vars the rest of the query
needs; the join computes the ORDER BY expression from those Vars, once
per joined row, though the scan already had the value.  Tom Lane
pointed at this in the 2023 junk-column thread: a fix that only munges
the final targetlist "leaves everything on the table as soon as there's
more than one level of plan involved".

Give such an IndexPath its own PathTarget: a copy of rel->reltarget
with each returnable ORDER BY expression appended, but only if the
query needs that expression above the scan (a top-level entry of the
final targetlist, which includes sort columns).  Most index paths keep
sharing rel->reltarget as before.  The appended expressions cost the
scan nothing (cost_index() now uses path_target_cost()).  Above the
scan, setrefs.c's fix_join_expr() already matches whole non-Var
expressions against the input targetlists before descending into them,
so the join's copy becomes a reference to the scan's output with no
setrefs change; and the join isn't charged for it either, since
path_target_cost() leaves out of a join path's target the entries an
input already emits (looking through Material and Memoize).
use_physical_tlist() must not replace such a scan's targetlist with a
physical one, which has only Vars; index_path_emits_orderby() tells it.

Scope: plain IndexScan paths on plain base relations only.  Not
IndexOnlyScan (no ORDER BY value plumbing), not appendrel children
(Append/MergeAppend build their own child targetlists, and the
MergeAppend-over-kNN case already works through the scan/join target),
and only ORDER BY expressions that pass index_orderby_returnable().
Because the join reads the value from the scan's output, an outer join
nulls it correctly with the rest of that side's columns.

Author: Greg Burd <greg@burd.me>
Discussion: https://postgr.es/m/8s5lT8erXzBMugXJQ6Wginbp_gc2B4hmCYhF0Q0GpnuK87eCAOcKs5K31AWCwraCNFIQt42wwkow_MPPubHg485MWv-zDBY4NL_JX9yJEEg=@burd.me
Discussion: https://postgr.es/m/2935085.1703806977@sss.pgh.pa.us
---
 src/backend/optimizer/path/costsize.c   |  12 +-
 src/backend/optimizer/plan/createplan.c |   9 ++
 src/backend/optimizer/util/pathnode.c   | 182 +++++++++++++++++++++++-
 src/include/optimizer/pathnode.h        |   3 +
 src/test/regress/expected/gist.out      |  91 ++++++++++++
 src/test/regress/sql/gist.sql           |  45 ++++++
 6 files changed, 336 insertions(+), 6 deletions(-)

diff --git a/src/backend/optimizer/path/costsize.c b/src/backend/optimizer/path/costsize.c
index 7bbddb8bee4..a844701b1d1 100644
--- a/src/backend/optimizer/path/costsize.c
+++ b/src/backend/optimizer/path/costsize.c
@@ -564,6 +564,7 @@ cost_index(IndexPath *path, PlannerInfo *root, double loop_count,
 	Cost		min_IO_cost,
 				max_IO_cost;
 	QualCost	qpqual_cost;
+	QualCost	tlist_cost;
 	Cost		cpu_per_tuple;
 	double		tuples_fetched;
 	double		pages_fetched;
@@ -799,9 +800,14 @@ cost_index(IndexPath *path, PlannerInfo *root, double loop_count,
 
 	cpu_run_cost += cpu_per_tuple * tuples_fetched;
 
-	/* tlist eval costs are paid per output row, not per tuple scanned */
-	startup_cost += path->path.pathtarget->cost.startup;
-	cpu_run_cost += path->path.pathtarget->cost.per_tuple * path->path.rows;
+	/*
+	 * tlist eval costs are paid per output row, not per tuple scanned.  An
+	 * ordering IndexScan doesn't pay for target entries it takes from its
+	 * ORDER BY values.
+	 */
+	tlist_cost = path_target_cost(root, (Path *) path, path->path.pathtarget);
+	startup_cost += tlist_cost.startup;
+	cpu_run_cost += tlist_cost.per_tuple * path->path.rows;
 
 	/* Adjust costing for parallelism, if used. */
 	if (path->path.parallel_workers > 0)
diff --git a/src/backend/optimizer/plan/createplan.c b/src/backend/optimizer/plan/createplan.c
index 1cc87997a9a..507c1c8fb35 100644
--- a/src/backend/optimizer/plan/createplan.c
+++ b/src/backend/optimizer/plan/createplan.c
@@ -934,6 +934,15 @@ use_physical_tlist(PlannerInfo *root, Path *path, int flags)
 			return false;
 	}
 
+	/*
+	 * Nor if a plain index scan path was asked to emit its ORDER BY values
+	 * (see index_path_orderby_target()); a physical tlist has only Vars, so
+	 * whatever needs them would compute them again.
+	 */
+	if (path->pathtype == T_IndexScan &&
+		index_path_emits_orderby((IndexPath *) path))
+		return false;
+
 	/*
 	 * For an index-only scan, the "physical tlist" is the index's indextlist.
 	 * We can only return that without a projection if all the index's columns
diff --git a/src/backend/optimizer/util/pathnode.c b/src/backend/optimizer/util/pathnode.c
index 303df571757..377491cf4f2 100644
--- a/src/backend/optimizer/util/pathnode.c
+++ b/src/backend/optimizer/util/pathnode.c
@@ -1070,6 +1070,120 @@ create_samplescan_path(PlannerInfo *root, RelOptInfo *rel, Relids required_outer
 	return pathnode;
 }
 
+/*
+ * index_path_orderby_target
+ *	  Choose the PathTarget for an ordering plain IndexScan path.
+ *
+ * Normally an IndexPath emits rel->reltarget, which holds only the Vars (and
+ * PlaceHolderVars) the rest of the query needs.  An ORDER BY expression is
+ * not in it: it is computed by whatever plan node first needs it.  When the
+ * scan is the top of the scan/join tree that is the scan itself, and
+ * setrefs.c lets it take the value from its ORDER BY values.  But when the
+ * scan is below a join, the join computes the expression from the Vars,
+ * once per joined row, though the scan already had the value.
+ *
+ * So if the scan can return any of its ORDER BY values (see
+ * index_orderby_returnable()), and the query needs that expression above
+ * the scan (it is in the final targetlist, which includes sort columns),
+ * give the path a copy of rel->reltarget with those expressions appended.
+ * They cost the scan nothing (path_target_cost() exempts them), and
+ * setrefs.c's fix_join_expr() matches them in the parent join's expressions,
+ * replacing its copy with a reference to the scan's output.
+ *
+ * Restricted to plain base relations: an appendrel child's paths are wrapped
+ * by Append/MergeAppend, which build their own targetlists, and the top-level
+ * scan/join target already handles the unjoined case.  Returns
+ * rel->reltarget unchanged when there's nothing to add, which is the common
+ * case, so most paths share it as before.
+ */
+static PathTarget *
+index_path_orderby_target(PlannerInfo *root, IndexOptInfo *index,
+						  List *indexorderbys, List *indexorderbycols)
+{
+	RelOptInfo *rel = index->rel;
+	PathTarget *target = NULL;
+	ListCell   *lo,
+			   *lcol;
+
+	if (indexorderbys == NIL || rel->reloptkind != RELOPT_BASEREL ||
+		root->processed_tlist == NIL)
+		return rel->reltarget;
+
+	forboth(lo, indexorderbys, lcol, indexorderbycols)
+	{
+		Expr	   *orderby = (Expr *) lfirst(lo);
+		bool		needed = false;
+		ListCell   *lt;
+
+		if (!index_orderby_returnable(index, lfirst_int(lcol), orderby))
+			continue;
+
+		/*
+		 * Only if some top-level targetlist entry is this expression; an
+		 * expression that merely contains it would not be rewritten, and
+		 * emitting the value would just widen the tuple.
+		 */
+		foreach(lt, root->processed_tlist)
+		{
+			TargetEntry *tle = lfirst_node(TargetEntry, lt);
+
+			if (orderby_tlist_match(tle->expr, orderby))
+			{
+				needed = true;
+				break;
+			}
+		}
+		if (!needed)
+			continue;
+
+		if (target == NULL)
+			target = copy_pathtarget(rel->reltarget);
+		add_column_to_pathtarget(target, orderby, 0);
+	}
+
+	if (target == NULL)
+		return rel->reltarget;
+
+	/* add_column_to_pathtarget doesn't maintain cost and width */
+	return set_pathtarget_cost_width(root, target);
+}
+
+/*
+ * index_path_emits_orderby
+ *	  Does this plain IndexScan path's target hold a value it takes from its
+ *	  ORDER BY values?
+ *
+ * use_physical_tlist() must not replace such a target with the scan's
+ * physical tlist, which has only Vars, or whatever needs the value would
+ * compute it again.
+ */
+bool
+index_path_emits_orderby(IndexPath *ipath)
+{
+	ListCell   *lc;
+
+	if (ipath->path.pathtype != T_IndexScan || ipath->indexorderbys == NIL)
+		return false;
+
+	foreach(lc, ipath->path.pathtarget->exprs)
+	{
+		Expr	   *expr = (Expr *) lfirst(lc);
+		ListCell   *lo,
+				   *lcol;
+
+		if (IsA(expr, Var))
+			continue;
+		forboth(lo, ipath->indexorderbys, lcol, ipath->indexorderbycols)
+		{
+			if (index_orderby_returnable(ipath->indexinfo, lfirst_int(lcol),
+										 (Expr *) lfirst(lo)) &&
+				orderby_tlist_match(expr, (Expr *) lfirst(lo)))
+				return true;
+		}
+	}
+	return false;
+}
+
 /*
  * create_index_path
  *	  Creates a path node for an index scan.
@@ -1109,7 +1223,9 @@ create_index_path(PlannerInfo *root,
 
 	pathnode->path.pathtype = indexonly ? T_IndexOnlyScan : T_IndexScan;
 	pathnode->path.parent = rel;
-	pathnode->path.pathtarget = rel->reltarget;
+	pathnode->path.pathtarget = indexonly ? rel->reltarget :
+		index_path_orderby_target(root, index, indexorderbys,
+								  indexorderbycols);
 	pathnode->path.param_info = get_baserel_parampathinfo(root, rel,
 														  required_outer);
 	pathnode->path.parallel_aware = false;
@@ -2675,24 +2791,84 @@ orderby_tlist_match(Expr *expr, Expr *orderby)
 		equal(lsecond(a->args), linitial(b->args));
 }
 
+/*
+ * join_target_cost
+ *	  path_target_cost() for a join path.
+ *
+ * A join takes any top-level target expression that one of its inputs
+ * already emits from that input's output instead of evaluating it again:
+ * setrefs.c's fix_join_expr() matches whole non-Var expressions against the
+ * input targetlists before descending into them.  Join inputs ordinarily
+ * emit only Vars and PlaceHolderVars, so this changes nothing; but an
+ * ordering IndexScan input may emit the ORDER BY values it returns (see
+ * index_path_orderby_target()).  Look through Material and Memoize, which
+ * pass their input's targetlist up unchanged.  We look only at the join's
+ * immediate inputs; the value can't reach a higher join without passing
+ * through this one's targetlist, which is the joinrel's and has only Vars.
+ */
+static QualCost
+join_target_cost(PlannerInfo *root, JoinPath *jpath, PathTarget *target)
+{
+	QualCost	cost = target->cost;
+	Path	   *inputs[2] = {jpath->outerjoinpath, jpath->innerjoinpath};
+	ListCell   *lc;
+
+	foreach(lc, target->exprs)
+	{
+		Node	   *expr = (Node *) lfirst(lc);
+
+		if (IsA(expr, Var) || IsA(expr, PlaceHolderVar))
+			continue;
+
+		for (int i = 0; i < 2; i++)
+		{
+			Path	   *input = inputs[i];
+
+			while (IsA(input, MaterialPath) || IsA(input, MemoizePath))
+				input = IsA(input, MaterialPath) ?
+					((MaterialPath *) input)->subpath :
+					((MemoizePath *) input)->subpath;
+
+			if (IsA(input, IndexPath) &&
+				index_path_emits_orderby((IndexPath *) input) &&
+				list_member(input->pathtarget->exprs, expr))
+			{
+				QualCost	ecost;
+
+				cost_qual_eval_node(&ecost, expr, root);
+				cost.startup -= ecost.startup;
+				cost.per_tuple -= ecost.per_tuple;
+				break;
+			}
+		}
+	}
+
+	return cost;
+}
+
 /*
  * path_target_cost
  *	  Return the cost 'path' pays to evaluate 'target'.
  *
- * Normally that's just target->cost.  But a plain IndexScan takes any
+ * Normally that's just target->cost.  A join path doesn't pay for entries
+ * an input already emits (see join_target_cost()).  And a plain IndexScan
+ * takes any
  * top-level target expression that is a returnable ORDER BY expression
  * (see orderby_tlist_match()) from its ORDER BY values instead of
  * evaluating it (setrefs.c rewrites it into a reference to them), so it
  * doesn't pay for those.  Each target entry
  * is counted at most once, as setrefs.c rewrites it once.
  */
-static QualCost
+QualCost
 path_target_cost(PlannerInfo *root, Path *path, PathTarget *target)
 {
 	QualCost	cost = target->cost;
 	IndexPath  *ipath;
 	ListCell   *lc;
 
+	if (IsA(path, NestPath) || IsA(path, MergePath) || IsA(path, HashPath))
+		return join_target_cost(root, (JoinPath *) path, target);
+
 	if (!IsA(path, IndexPath) || path->pathtype != T_IndexScan)
 		return cost;
 	ipath = (IndexPath *) path;
diff --git a/src/include/optimizer/pathnode.h b/src/include/optimizer/pathnode.h
index fad229234fd..8fcfe60aba3 100644
--- a/src/include/optimizer/pathnode.h
+++ b/src/include/optimizer/pathnode.h
@@ -229,6 +229,9 @@ extern void sort_pathlist_by_cost(List *pathlist);
 extern bool index_orderby_returnable(IndexOptInfo *index, int indexcol,
 									 Expr *orderby);
 extern bool orderby_tlist_match(Expr *expr, Expr *orderby);
+extern QualCost path_target_cost(PlannerInfo *root, Path *path,
+								 PathTarget *target);
+extern bool index_path_emits_orderby(IndexPath *ipath);
 extern ProjectionPath *create_projection_path(PlannerInfo *root,
 											  RelOptInfo *rel,
 											  Path *subpath,
diff --git a/src/test/regress/expected/gist.out b/src/test/regress/expected/gist.out
index df37c8096d1..4fc2b084df1 100644
--- a/src/test/regress/expected/gist.out
+++ b/src/test/regress/expected/gist.out
@@ -631,6 +631,97 @@ reset enable_seqscan;
 reset enable_bitmapscan;
 reset enable_indexonlyscan;
 drop table gist_commute;
+-- An ordering Index Scan below a join emits the ORDER BY value it returns,
+-- so the join above reads it instead of evaluating the expression again.
+-- The counting function shows the expression is evaluated zero times.
+create sequence gist_join_cnt;
+create function gist_join_pt(point) returns point
+  language plpgsql immutable strict cost 1000
+  as $$ begin perform nextval('public.gist_join_cnt'); return $1; end $$;
+create temp table gist_join_fact as
+  select g as id, point(g % 101, g % 103) as p from generate_series(1, 1000) g;
+create index on gist_join_fact using gist (gist_join_pt(p));
+create temp table gist_join_dim as select g as id from generate_series(1, 1000, 2) g;
+create index on gist_join_dim (id);
+vacuum analyze gist_join_fact;
+vacuum analyze gist_join_dim;
+set enable_seqscan = off;
+set enable_bitmapscan = off;
+set enable_indexonlyscan = off;
+set enable_hashjoin = off;
+set enable_mergejoin = off;
+set enable_sort = off;
+explain (verbose, costs off)
+select f.id, gist_join_pt(f.p) <-> point(5,5) as d
+  from gist_join_fact f join gist_join_dim j on j.id = f.id
+  order by gist_join_pt(f.p) <-> point(5,5) limit 3;
+                                         QUERY PLAN                                         
+--------------------------------------------------------------------------------------------
+ Limit
+   Output: f.id, ((gist_join_pt(f.p) <-> '(5,5)'::point))
+   ->  Nested Loop
+         Output: f.id, ((gist_join_pt(f.p) <-> '(5,5)'::point))
+         ->  Index Scan using gist_join_fact_gist_join_pt_p_idx on pg_temp.gist_join_fact f
+               Output: f.id, f.p, ((gist_join_pt(f.p) <-> '(5,5)'::point))
+               Order By: (gist_join_pt(f.p) <-> '(5,5)'::point)
+               Order By Values Used: 1
+         ->  Index Scan using gist_join_dim_id_idx on pg_temp.gist_join_dim j
+               Output: j.id
+               Index Cond: (j.id = f.id)
+(11 rows)
+
+select setval('gist_join_cnt', 1, false);
+ setval 
+--------
+      1
+(1 row)
+
+select f.id, gist_join_pt(f.p) <-> point(5,5) as d
+  from gist_join_fact f join gist_join_dim j on j.id = f.id
+  order by gist_join_pt(f.p) <-> point(5,5) limit 3;
+ id  |         d          
+-----+--------------------
+   5 |                  0
+ 107 | 1.4142135623730951
+ 209 | 2.8284271247461903
+(3 rows)
+
+select nextval('gist_join_cnt') - 1 as calls_during_join;
+ calls_during_join 
+-------------------
+                 0
+(1 row)
+
+-- Nothing extra is emitted when the query doesn't need the value above the
+-- scan.
+explain (verbose, costs off)
+select f.id from gist_join_fact f join gist_join_dim j on j.id = f.id
+  order by gist_join_pt(f.p) <-> point(5,5) limit 3;
+                                         QUERY PLAN                                         
+--------------------------------------------------------------------------------------------
+ Limit
+   Output: f.id, ((gist_join_pt(f.p) <-> '(5,5)'::point))
+   ->  Nested Loop
+         Output: f.id, ((gist_join_pt(f.p) <-> '(5,5)'::point))
+         ->  Index Scan using gist_join_fact_gist_join_pt_p_idx on pg_temp.gist_join_fact f
+               Output: f.id, f.p, ((gist_join_pt(f.p) <-> '(5,5)'::point))
+               Order By: (gist_join_pt(f.p) <-> '(5,5)'::point)
+               Order By Values Used: 1
+         ->  Index Scan using gist_join_dim_id_idx on pg_temp.gist_join_dim j
+               Output: j.id
+               Index Cond: (j.id = f.id)
+(11 rows)
+
+reset enable_seqscan;
+reset enable_bitmapscan;
+reset enable_indexonlyscan;
+reset enable_hashjoin;
+reset enable_mergejoin;
+reset enable_sort;
+drop table gist_join_fact;
+drop table gist_join_dim;
+drop function gist_join_pt(point);
+drop sequence gist_join_cnt;
 -- Test that an index-only scan deforms the tuple it reconstructs with the
 -- descriptor the AM formed it with, not the scan slot's descriptor.
 create temp table gist_ios_tupdesc (a inet, r numrange);
diff --git a/src/test/regress/sql/gist.sql b/src/test/regress/sql/gist.sql
index 87492a11df4..9e7bc0fd176 100644
--- a/src/test/regress/sql/gist.sql
+++ b/src/test/regress/sql/gist.sql
@@ -309,6 +309,51 @@ reset enable_bitmapscan;
 reset enable_indexonlyscan;
 drop table gist_commute;
 
+-- An ordering Index Scan below a join emits the ORDER BY value it returns,
+-- so the join above reads it instead of evaluating the expression again.
+-- The counting function shows the expression is evaluated zero times.
+create sequence gist_join_cnt;
+create function gist_join_pt(point) returns point
+  language plpgsql immutable strict cost 1000
+  as $$ begin perform nextval('public.gist_join_cnt'); return $1; end $$;
+create temp table gist_join_fact as
+  select g as id, point(g % 101, g % 103) as p from generate_series(1, 1000) g;
+create index on gist_join_fact using gist (gist_join_pt(p));
+create temp table gist_join_dim as select g as id from generate_series(1, 1000, 2) g;
+create index on gist_join_dim (id);
+vacuum analyze gist_join_fact;
+vacuum analyze gist_join_dim;
+set enable_seqscan = off;
+set enable_bitmapscan = off;
+set enable_indexonlyscan = off;
+set enable_hashjoin = off;
+set enable_mergejoin = off;
+set enable_sort = off;
+explain (verbose, costs off)
+select f.id, gist_join_pt(f.p) <-> point(5,5) as d
+  from gist_join_fact f join gist_join_dim j on j.id = f.id
+  order by gist_join_pt(f.p) <-> point(5,5) limit 3;
+select setval('gist_join_cnt', 1, false);
+select f.id, gist_join_pt(f.p) <-> point(5,5) as d
+  from gist_join_fact f join gist_join_dim j on j.id = f.id
+  order by gist_join_pt(f.p) <-> point(5,5) limit 3;
+select nextval('gist_join_cnt') - 1 as calls_during_join;
+-- Nothing extra is emitted when the query doesn't need the value above the
+-- scan.
+explain (verbose, costs off)
+select f.id from gist_join_fact f join gist_join_dim j on j.id = f.id
+  order by gist_join_pt(f.p) <-> point(5,5) limit 3;
+reset enable_seqscan;
+reset enable_bitmapscan;
+reset enable_indexonlyscan;
+reset enable_hashjoin;
+reset enable_mergejoin;
+reset enable_sort;
+drop table gist_join_fact;
+drop table gist_join_dim;
+drop function gist_join_pt(point);
+drop sequence gist_join_cnt;
+
 -- Test that an index-only scan deforms the tuple it reconstructs with the
 -- descriptor the AM formed it with, not the scan slot's descriptor.
 create temp table gist_ios_tupdesc (a inet, r numrange);
-- 
2.50.1

