From 69dca9e1b4b6b3f3c360e3e694b87c67192e961b Mon Sep 17 00:00:00 2001
From: Greg Burd <greg@burd.me>
Date: Thu, 8 Oct 2026 07:55:02 -0400
Subject: [PATCH v7 6/6] Carry index-computed values up through joins and
 Appends

The previous commit lets an index-only scan beat a seq scan when it
reads an expression the query's targetlist needs, but only when the
scan's rel is the only one in the query and the targetlist entry is
that exact expression.  Below a join the scan isn't chosen, and the
expression is computed above it as before.  Nor is it chosen when the
expression is part of a larger one, such as expensive(i) + 1, or when
the table is partitioned.

Generalize the extra values an index path may emit, which 0003 added
for an ordering IndexScan's ORDER BY values and the previous commit for
an index-only scan's expression columns:

* A join emits what its inputs emit, so the value reaches the scan/join
  target from any depth.  fix_join_expr() already matches whole
  expressions in an input's targetlist, and path_target_cost() now
  credits them wherever they appear in an expression, as setrefs.c
  replaces them.  Material and Memoize share their input's target.

* An index-only scan emits a returnable index expression wherever it
  appears in the targetlist, including ORDER BY columns.  Its qpquals
  aren't charged for one either, since set_indexonlyscan_references()
  rewrites them too.

* For a partitioned table whose partitions all have the expression
  index, an unordered Append over child paths that all emit the value
  emits it too.  create_append_child_plan() puts each child's values in
  the Append's order, translating them to the child, and drops any
  extra values a child emits for an Append that doesn't.

* Hash and merge joins look only at each input's cheapest path, which
  rarely emits the value.  Also try each input's cheapest path that
  does, and skip add_path_precheck() for paths over such an input, as
  the precheck can't see the advantage.

add_path() and add_partial_path() compare the extra values like
pathkeys: a path that emits a superset of another's is better, so it
can dominate it, but not the other way round.  The previous commit's
rule kept both whenever the targets differed, which kept, for example,
a seq scan that a cheaper, better-ordered kNN path dominates.

The cost is planning time: every join path over an input with an extra
value is kept alongside the one it would otherwise have replaced.  So a
value is emitted only if it costs more than 10 times cpu_operator_cost
per evaluation, the same test make_sort_input_target() uses to decide
an expression is worth postponing past a sort.  Queries without such an
expression plan as before.

A value from the nullable side of an outer join is never used above
it: the query's Vars carry varnullingrels the index's don't, so they
don't match.  Partial paths emit only parallel-safe values.  An ordered
or parallel Append still computes the expression above it.

Author: Greg Burd <greg@burd.me>
Discussion: https://postgr.es/m/8s5lT8erXzBMugXJQ6Wginbp_gc2B4hmCYhF0Q0GpnuK87eCAOcKs5K31AWCwraCNFIQt42wwkow_MPPubHg485MWv-zDBY4NL_JX9yJEEg=@burd.me
Discussion: https://postgr.es/m/356d5811-d8f0-4c29-a601-b61334337b11@iki.fi
---
 src/backend/optimizer/path/allpaths.c   | 104 ++++
 src/backend/optimizer/path/costsize.c   |  47 +-
 src/backend/optimizer/path/joinpath.c   | 124 ++++-
 src/backend/optimizer/plan/createplan.c |  70 ++-
 src/backend/optimizer/plan/planner.c    |  11 +-
 src/backend/optimizer/util/pathnode.c   | 667 +++++++++++++++++-------
 src/include/nodes/pathnodes.h           |   5 +
 src/include/optimizer/pathnode.h        |   7 +-
 src/test/regress/expected/gist.out      | 158 +++++-
 src/test/regress/sql/gist.sql           |  75 ++-
 10 files changed, 1046 insertions(+), 222 deletions(-)

diff --git a/src/backend/optimizer/path/allpaths.c b/src/backend/optimizer/path/allpaths.c
index d52a6d40505..417f9b1c544 100644
--- a/src/backend/optimizer/path/allpaths.c
+++ b/src/backend/optimizer/path/allpaths.c
@@ -103,6 +103,8 @@ static void set_rel_pathlist(PlannerInfo *root, RelOptInfo *rel,
 static void set_plain_rel_size(PlannerInfo *root, RelOptInfo *rel,
 							   RangeTblEntry *rte);
 static void create_plain_partial_paths(PlannerInfo *root, RelOptInfo *rel);
+static void add_extras_append_path(PlannerInfo *root, RelOptInfo *rel,
+								   List *live_childrels);
 static void set_rel_consider_parallel(PlannerInfo *root, RelOptInfo *rel,
 									  RangeTblEntry *rte);
 static void set_plain_rel_pathlist(PlannerInfo *root, RelOptInfo *rel,
@@ -1395,6 +1397,100 @@ set_grouped_rel_pathlist(PlannerInfo *root, RelOptInfo *rel)
 }
 
 
+/*
+ * add_extras_append_path
+ *	  Build an unordered, unparameterized Append over child paths that all
+ *	  emit the same extra values, if there are any worth emitting.
+ *
+ * A partitioned table's index on an expression is an index on that
+ * expression in every partition, so each child may have an index-only scan
+ * that reads it (see indexonly_path_target()).  The Append can pass such a
+ * value up, as a join does, but only if every child emits it: they must all
+ * return the Append's targetlist.  The candidates are the parent's
+ * partitioned indexes' expressions that the query's targetlist uses; for
+ * each, translated to the child, we take the child's cheapest path that
+ * emits it.  create_append_child_plan() puts each child's values in the
+ * Append's order.
+ *
+ * Only an unordered, non-partial Append is built this way; an ordered or
+ * parallel Append still computes the expression above it.
+ */
+static void
+add_extras_append_path(PlannerInfo *root, RelOptInfo *rel,
+					   List *live_childrels)
+{
+	List	   *extras = NIL;
+	List	   *tlist_exprs;
+	AppendPathInput input = {0};
+	PathTarget *target;
+	AppendPath *apath;
+	ListCell   *lc;
+
+	if (rel->reloptkind != RELOPT_BASEREL || live_childrels == NIL ||
+		root->processed_tlist == NIL)
+		return;
+
+	tlist_exprs = get_tlist_exprs(root->processed_tlist, true);
+	foreach(lc, rel->indexlist)
+	{
+		IndexOptInfo *index = (IndexOptInfo *) lfirst(lc);
+		ListCell   *le;
+
+		foreach(le, index->indexprs)
+		{
+			Node	   *expr = (Node *) lfirst(le);
+			ListCell   *lt;
+
+			if (!expr_worth_emitting(root, expr))
+				continue;
+			foreach(lt, tlist_exprs)
+			{
+				if (expr_contains(lfirst(lt), expr))
+				{
+					extras = list_append_unique(extras, expr);
+					break;
+				}
+			}
+		}
+	}
+	if (extras == NIL)
+		return;
+
+	foreach(lc, live_childrels)
+	{
+		RelOptInfo *childrel = (RelOptInfo *) lfirst(lc);
+		List	   *child_extras;
+		Path	   *best = NULL;
+		ListCell   *lp;
+
+		child_extras = (List *)
+			adjust_appendrel_attrs_multilevel(root, (Node *) extras,
+											  childrel, childrel->top_parent);
+		foreach(lp, childrel->pathlist)
+		{
+			Path	   *path = (Path *) lfirst(lp);
+
+			if (path->param_info == NULL &&
+				list_difference(child_extras, path_extra_exprs(path)) == NIL)
+			{
+				best = path;	/* pathlist is sorted by total cost */
+				break;
+			}
+		}
+		if (best == NULL)
+			return;
+		accumulate_append_subpath(best, &input.subpaths, NULL,
+								  &input.child_append_relid_sets);
+	}
+
+	apath = create_append_path(root, rel, input, NIL, NULL, 0, false, -1);
+	target = copy_pathtarget(rel->reltarget);
+	foreach(lc, extras)
+		add_new_column_to_pathtarget(target, (Expr *) lfirst(lc));
+	apath->path.pathtarget = set_pathtarget_cost_width(root, target);
+	add_path(rel, (Path *) apath);
+}
+
 /*
  * add_paths_to_append_rel
  *		Generate paths for the given append relation given the set of non-dummy
@@ -1619,6 +1715,14 @@ add_paths_to_append_rel(PlannerInfo *root, RelOptInfo *rel,
 												  NIL, NULL, 0, false,
 												  -1));
 
+	/*
+	 * Also consider an Append that emits extra values (see
+	 * path_extra_exprs()) its children read from their indexes, so a join
+	 * above it doesn't compute them.
+	 */
+	if (unparameterized_valid)
+		add_extras_append_path(root, rel, live_childrels);
+
 	/* build an AppendPath for the cheap startup paths, if valid */
 	if (startup_valid)
 		add_path(rel, (Path *) create_append_path(root, rel, startup,
diff --git a/src/backend/optimizer/path/costsize.c b/src/backend/optimizer/path/costsize.c
index a844701b1d1..4bd912d0811 100644
--- a/src/backend/optimizer/path/costsize.c
+++ b/src/backend/optimizer/path/costsize.c
@@ -795,6 +795,13 @@ cost_index(IndexPath *path, PlannerInfo *root, double loop_count,
 	 */
 	cost_qual_eval(&qpqual_cost, qpquals, root);
 
+	/*
+	 * An index-only scan reads returnable index expressions in its quals from
+	 * the index, as it does in its targetlist; see path_target_cost().
+	 */
+	if (indexonly)
+		qpqual_cost = indexonly_qual_cost(root, path, qpquals, qpqual_cost);
+
 	startup_cost += qpqual_cost.startup;
 	cpu_per_tuple = cpu_tuple_cost + qpqual_cost.per_tuple;
 
@@ -802,8 +809,8 @@ cost_index(IndexPath *path, PlannerInfo *root, double loop_count,
 
 	/*
 	 * tlist eval costs are paid per output row, not per tuple scanned.  An
-	 * ordering IndexScan doesn't pay for target entries it takes from its
-	 * ORDER BY values.
+	 * index path doesn't pay for target entries it reads from the index; see
+	 * path_target_cost().
 	 */
 	tlist_cost = path_target_cost(root, (Path *) path, path->path.pathtarget);
 	startup_cost += tlist_cost.startup;
@@ -3652,9 +3659,15 @@ final_cost_nestloop(PlannerInfo *root, NestPath *path,
 	cpu_per_tuple = cpu_tuple_cost + restrict_qual_cost.per_tuple;
 	run_cost += cpu_per_tuple * ntuples;
 
-	/* tlist eval costs are paid per output row, not per tuple scanned */
-	startup_cost += path->jpath.path.pathtarget->cost.startup;
-	run_cost += path->jpath.path.pathtarget->cost.per_tuple * path->jpath.path.rows;
+	/*
+	 * tlist eval costs are paid per output row, not per tuple scanned.  The
+	 * join may also emit extra values its inputs emit (see
+	 * join_path_target()), but it doesn't evaluate those, so charge only for
+	 * the joinrel's reltarget.
+	 */
+	startup_cost += path->jpath.path.parent->reltarget->cost.startup;
+	run_cost += path->jpath.path.parent->reltarget->cost.per_tuple *
+		path->jpath.path.rows;
 
 	path->jpath.path.startup_cost = startup_cost;
 	path->jpath.path.total_cost = startup_cost + run_cost;
@@ -4242,9 +4255,15 @@ final_cost_mergejoin(PlannerInfo *root, MergePath *path,
 	cpu_per_tuple = cpu_tuple_cost + qp_qual_cost.per_tuple;
 	run_cost += cpu_per_tuple * mergejointuples;
 
-	/* tlist eval costs are paid per output row, not per tuple scanned */
-	startup_cost += path->jpath.path.pathtarget->cost.startup;
-	run_cost += path->jpath.path.pathtarget->cost.per_tuple * path->jpath.path.rows;
+	/*
+	 * tlist eval costs are paid per output row, not per tuple scanned.  The
+	 * join may also emit extra values its inputs emit (see
+	 * join_path_target()), but it doesn't evaluate those, so charge only for
+	 * the joinrel's reltarget.
+	 */
+	startup_cost += path->jpath.path.parent->reltarget->cost.startup;
+	run_cost += path->jpath.path.parent->reltarget->cost.per_tuple *
+		path->jpath.path.rows;
 
 	path->jpath.path.startup_cost = startup_cost;
 	path->jpath.path.total_cost = startup_cost + run_cost;
@@ -4761,9 +4780,15 @@ final_cost_hashjoin(PlannerInfo *root, HashPath *path,
 		run_cost += cpu_per_tuple * hashjointuples;
 	}
 
-	/* tlist eval costs are paid per output row, not per tuple scanned */
-	startup_cost += path->jpath.path.pathtarget->cost.startup;
-	run_cost += path->jpath.path.pathtarget->cost.per_tuple * path->jpath.path.rows;
+	/*
+	 * tlist eval costs are paid per output row, not per tuple scanned.  The
+	 * join may also emit extra values its inputs emit (see
+	 * join_path_target()), but it doesn't evaluate those, so charge only for
+	 * the joinrel's reltarget.
+	 */
+	startup_cost += path->jpath.path.parent->reltarget->cost.startup;
+	run_cost += path->jpath.path.parent->reltarget->cost.per_tuple *
+		path->jpath.path.rows;
 
 	path->jpath.path.startup_cost = startup_cost;
 	path->jpath.path.total_cost = startup_cost + run_cost;
diff --git a/src/backend/optimizer/path/joinpath.c b/src/backend/optimizer/path/joinpath.c
index dfd08e7aeb1..3be2a9c24f2 100644
--- a/src/backend/optimizer/path/joinpath.c
+++ b/src/backend/optimizer/path/joinpath.c
@@ -45,6 +45,7 @@ join_path_setup_hook_type join_path_setup_hook = NULL;
 #define PATH_PARAM_BY_REL(path, rel)	\
 	(PATH_PARAM_BY_REL_SELF(path, rel) || PATH_PARAM_BY_PARENT(path, rel))
 
+static Path *cheapest_extras_path(RelOptInfo *rel);
 static void try_partial_mergejoin_path(PlannerInfo *root,
 									   RelOptInfo *joinrel,
 									   Path *outer_path,
@@ -149,6 +150,8 @@ add_paths_to_joinrel(PlannerInfo *root,
 	extra.restrictlist = restrictlist;
 	extra.mergeclause_list = NIL;
 	extra.sjinfo = sjinfo;
+	extra.extras_outer = cheapest_extras_path(outerrel);
+	extra.extras_inner = cheapest_extras_path(innerrel);
 	extra.param_source_rels = NULL;
 	extra.pgs_mask = joinrel->pgs_mask;
 
@@ -872,6 +875,51 @@ get_memoize_path(PlannerInfo *root, RelOptInfo *innerrel,
 	return NULL;
 }
 
+/*
+ * inputs_emit_extras
+ *	  Does either input path emit values beyond its reltarget?
+ *
+ * A join over such an input emits them too (see join_path_target()), which
+ * add_path() counts in its favor like better pathkeys.  add_path_precheck()
+ * can't know that before the path exists, so skip the precheck for it.
+ */
+static inline bool
+inputs_emit_extras(Path *outer_path, Path *inner_path)
+{
+	/* cheap test first: most paths share their rel's reltarget */
+	return (outer_path->pathtarget != outer_path->parent->reltarget &&
+			path_emits_extras(outer_path)) ||
+		(inner_path->pathtarget != inner_path->parent->reltarget &&
+		 path_emits_extras(inner_path));
+}
+
+/*
+ * cheapest_extras_path
+ *	  The cheapest unparameterized path of 'rel' that emits values beyond its
+ *	  reltarget (see path_extra_exprs()), or NULL.
+ *
+ * Such a path is kept by add_path() but is rarely the rel's cheapest, and
+ * hash and sort-merge joins look only at the cheapest.  It is worth trying
+ * as well: a join over it doesn't pay for evaluating those values.
+ * pathlist is sorted by total cost, so the first one found is the cheapest.
+ */
+static Path *
+cheapest_extras_path(RelOptInfo *rel)
+{
+	ListCell   *lc;
+
+	foreach(lc, rel->pathlist)
+	{
+		Path	   *path = (Path *) lfirst(lc);
+
+		if (path->param_info == NULL && path != rel->cheapest_total_path &&
+			path->pathtarget != rel->reltarget &&
+			path_emits_extras(path))
+			return path;
+	}
+	return NULL;
+}
+
 /*
  * try_nestloop_path
  *	  Consider a nestloop join path; if it appears useful, push it into
@@ -970,7 +1018,8 @@ try_nestloop_path(PlannerInfo *root,
 						  nestloop_subtype | PGS_CONSIDER_NONPARTIAL,
 						  outer_path, inner_path, extra);
 
-	if (add_path_precheck(joinrel, workspace.disabled_nodes,
+	if (inputs_emit_extras(outer_path, inner_path) ||
+		add_path_precheck(joinrel, workspace.disabled_nodes,
 						  workspace.startup_cost, workspace.total_cost,
 						  pathkeys, required_outer))
 	{
@@ -1055,7 +1104,8 @@ try_partial_nestloop_path(PlannerInfo *root,
 	 */
 	initial_cost_nestloop(root, &workspace, jointype, nestloop_subtype,
 						  outer_path, inner_path, extra);
-	if (!add_partial_path_precheck(joinrel, workspace.disabled_nodes,
+	if (!inputs_emit_extras(outer_path, inner_path) &&
+		!add_partial_path_precheck(joinrel, workspace.disabled_nodes,
 								   workspace.startup_cost,
 								   workspace.total_cost, pathkeys))
 		return;
@@ -1163,7 +1213,8 @@ try_mergejoin_path(PlannerInfo *root,
 						   outer_presorted_keys,
 						   extra);
 
-	if (add_path_precheck(joinrel, workspace.disabled_nodes,
+	if (inputs_emit_extras(outer_path, inner_path) ||
+		add_path_precheck(joinrel, workspace.disabled_nodes,
 						  workspace.startup_cost, workspace.total_cost,
 						  pathkeys, required_outer))
 	{
@@ -1245,7 +1296,8 @@ try_partial_mergejoin_path(PlannerInfo *root,
 						   outer_presorted_keys,
 						   extra);
 
-	if (!add_partial_path_precheck(joinrel, workspace.disabled_nodes,
+	if (!inputs_emit_extras(outer_path, inner_path) &&
+		!add_partial_path_precheck(joinrel, workspace.disabled_nodes,
 								   workspace.startup_cost,
 								   workspace.total_cost, pathkeys))
 		return;
@@ -1317,7 +1369,8 @@ try_hashjoin_path(PlannerInfo *root,
 	initial_cost_hashjoin(root, &workspace, jointype, hashclauses,
 						  outer_path, inner_path, extra, false);
 
-	if (add_path_precheck(joinrel, workspace.disabled_nodes,
+	if (inputs_emit_extras(outer_path, inner_path) ||
+		add_path_precheck(joinrel, workspace.disabled_nodes,
 						  workspace.startup_cost, workspace.total_cost,
 						  NIL, required_outer))
 	{
@@ -1378,7 +1431,8 @@ try_partial_hashjoin_path(PlannerInfo *root,
 	 */
 	initial_cost_hashjoin(root, &workspace, jointype, hashclauses,
 						  outer_path, inner_path, extra, parallel_hash);
-	if (!add_partial_path_precheck(joinrel, workspace.disabled_nodes,
+	if (!inputs_emit_extras(outer_path, inner_path) &&
+		!add_partial_path_precheck(joinrel, workspace.disabled_nodes,
 								   workspace.startup_cost,
 								   workspace.total_cost, NIL))
 		return;
@@ -1419,6 +1473,8 @@ sort_inner_and_outer(PlannerInfo *root,
 {
 	Path	   *outer_path;
 	Path	   *inner_path;
+	Path	   *extras_outer;
+	Path	   *extras_inner;
 	Path	   *cheapest_partial_outer = NULL;
 	Path	   *cheapest_safe_inner = NULL;
 	List	   *all_pathkeys;
@@ -1453,6 +1509,8 @@ sort_inner_and_outer(PlannerInfo *root,
 	if (PATH_PARAM_BY_REL(outer_path, innerrel) ||
 		PATH_PARAM_BY_REL(inner_path, outerrel))
 		return;
+	extras_outer = extra->extras_outer;
+	extras_inner = extra->extras_inner;
 
 	/*
 	 * If the joinrel is parallel-safe, we may be able to consider a partial
@@ -1561,6 +1619,16 @@ sort_inner_and_outer(PlannerInfo *root,
 						   extra,
 						   false);
 
+		/* Likewise with inputs that emit extra values; see hash join */
+		if (extras_outer != NULL)
+			try_mergejoin_path(root, joinrel, extras_outer, inner_path,
+							   merge_pathkeys, cur_mergeclauses,
+							   outerkeys, innerkeys, jointype, extra, false);
+		if (extras_inner != NULL)
+			try_mergejoin_path(root, joinrel, outer_path, extras_inner,
+							   merge_pathkeys, cur_mergeclauses,
+							   outerkeys, innerkeys, jointype, extra, false);
+
 		/*
 		 * If we have partial outer and parallel safe inner path then try
 		 * partial mergejoin path.
@@ -1845,6 +1913,7 @@ match_unsorted_outer(PlannerInfo *root,
 	bool		useallclauses;
 	Path	   *inner_cheapest_total = innerrel->cheapest_total_path;
 	Path	   *matpath = NULL;
+	Path	   *extras_matpath = NULL;
 	ListCell   *lc1;
 
 	/*
@@ -1921,6 +1990,20 @@ match_unsorted_outer(PlannerInfo *root,
 			!ExecMaterializesOutput(inner_cheapest_total->pathtype))
 			matpath = (Path *)
 				create_material_path(innerrel, inner_cheapest_total, true);
+
+		if ((extra->pgs_mask &
+			 (PGS_NESTLOOP_MATERIALIZE | PGS_CONSIDER_NONPARTIAL)) ==
+			(PGS_NESTLOOP_MATERIALIZE | PGS_CONSIDER_NONPARTIAL) &&
+			inner_cheapest_total != NULL)
+		{
+			Path	   *extras_inner = extra->extras_inner;
+
+			if (extras_inner != NULL &&
+				!PATH_PARAM_BY_REL(extras_inner, outerrel) &&
+				!ExecMaterializesOutput(extras_inner->pathtype))
+				extras_matpath = (Path *)
+					create_material_path(innerrel, extras_inner, true);
+		}
 	}
 
 	foreach(lc1, outerrel->pathlist)
@@ -1994,6 +2077,17 @@ match_unsorted_outer(PlannerInfo *root,
 								  jointype,
 								  PGS_NESTLOOP_MATERIALIZE,
 								  extra);
+
+			/* And of an inner path that emits extra values; see hash join */
+			if (extras_matpath != NULL)
+				try_nestloop_path(root,
+								  joinrel,
+								  outerpath,
+								  extras_matpath,
+								  merge_pathkeys,
+								  jointype,
+								  PGS_NESTLOOP_MATERIALIZE,
+								  extra);
 		}
 
 		/* Can't do anything else if inner rel is parameterized by outer */
@@ -2283,6 +2377,24 @@ hash_inner_and_outer(PlannerInfo *root,
 							  jointype,
 							  extra);
 
+		/*
+		 * Also try an input that emits values the query needs above the join,
+		 * on either side; see cheapest_extras_path().
+		 */
+		{
+			Path	   *extras_outer = extra->extras_outer;
+			Path	   *extras_inner = extra->extras_inner;
+
+			if (extras_outer != NULL)
+				try_hashjoin_path(root, joinrel,
+								  extras_outer, cheapest_total_inner,
+								  hashclauses, jointype, extra);
+			if (extras_inner != NULL)
+				try_hashjoin_path(root, joinrel,
+								  cheapest_total_outer, extras_inner,
+								  hashclauses, jointype, extra);
+		}
+
 		foreach(lc1, outerrel->cheapest_parameterized_paths)
 		{
 			Path	   *outerpath = (Path *) lfirst(lc1);
diff --git a/src/backend/optimizer/plan/createplan.c b/src/backend/optimizer/plan/createplan.c
index 507c1c8fb35..def3923c5d5 100644
--- a/src/backend/optimizer/plan/createplan.c
+++ b/src/backend/optimizer/plan/createplan.c
@@ -24,6 +24,7 @@
 #include "nodes/extensible.h"
 #include "nodes/makefuncs.h"
 #include "nodes/nodeFuncs.h"
+#include "optimizer/appendinfo.h"
 #include "optimizer/clauses.h"
 #include "optimizer/cost.h"
 #include "optimizer/optimizer.h"
@@ -83,6 +84,8 @@ static Plan *create_gating_plan(PlannerInfo *root, Path *path, Plan *plan,
 								List *gating_quals);
 static Plan *create_join_plan(PlannerInfo *root, JoinPath *best_path);
 static bool mark_async_capable_plan(Plan *plan, Path *path);
+static Plan *create_append_child_plan(PlannerInfo *root, Path *apath,
+									  Path *subpath);
 static Plan *create_append_plan(PlannerInfo *root, AppendPath *best_path,
 								int flags);
 static Plan *create_merge_append_plan(PlannerInfo *root, MergeAppendPath *best_path,
@@ -935,12 +938,12 @@ use_physical_tlist(PlannerInfo *root, Path *path, int flags)
 	}
 
 	/*
-	 * Nor if a plain index scan path was asked to emit its ORDER BY values
-	 * (see index_path_orderby_target()); a physical tlist has only Vars, so
-	 * whatever needs them would compute them again.
+	 * Nor if the path emits extra values (see path_extra_exprs()), such as an
+	 * index scan's ORDER BY values or an index-only scan's expression
+	 * columns; a physical tlist wouldn't put them where whatever needs them
+	 * expects.
 	 */
-	if (path->pathtype == T_IndexScan &&
-		index_path_emits_orderby((IndexPath *) path))
+	if (path_emits_extras(path))
 		return false;
 
 	/*
@@ -1213,6 +1216,59 @@ mark_async_capable_plan(Plan *plan, Path *path)
 	return true;
 }
 
+/*
+ * create_append_child_plan
+ *	  Build the plan for a child of an Append or MergeAppend.
+ *
+ * Every child must return the Append's tlist: its rel's reltarget, plus any
+ * extra values the Append emits (see add_extras_append_path()), translated
+ * to the child.  A child path may emit extra values of its own beyond its
+ * reltarget (see path_extra_exprs()), in some order; keep those the Append
+ * emits, in the Append's order, and drop the rest.
+ */
+static Plan *
+create_append_child_plan(PlannerInfo *root, Path *apath, Path *subpath)
+{
+	Plan	   *subplan = create_plan_recurse(root, subpath, CP_EXACT_TLIST);
+	List	   *append_extras = path_extra_exprs(apath);
+	List	   *child_extras;
+	List	   *tlist;
+	int			nrel;
+	ListCell   *lc;
+
+	if (!path_emits_extras(subpath))
+	{
+		Assert(append_extras == NIL);
+		return subplan;
+	}
+
+	nrel = list_length(subpath->parent->reltarget->exprs);
+	tlist = list_copy_head(subplan->targetlist, nrel);
+	child_extras = (List *)
+		adjust_appendrel_attrs_multilevel(root, (Node *) append_extras,
+										  subpath->parent,
+										  apath->parent);
+	foreach(lc, child_extras)
+	{
+		TargetEntry *tle = tlist_member((Expr *) lfirst(lc),
+										subplan->targetlist);
+
+		if (tle == NULL)
+			elog(ERROR, "Append child doesn't emit an Append's extra value");
+		tlist = lappend(tlist, makeTargetEntry(tle->expr,
+											   list_length(tlist) + 1,
+											   NULL, false));
+	}
+	if (tlist_same_exprs(tlist, subplan->targetlist))
+		return subplan;
+	if (is_projection_capable_plan(subplan))
+	{
+		subplan->targetlist = tlist;
+		return subplan;
+	}
+	return inject_projection_plan(subplan, tlist, subplan->parallel_safe);
+}
+
 /*
  * create_append_plan
  *	  Create an Append plan for 'best_path' and (recursively) plans
@@ -1314,7 +1370,7 @@ create_append_plan(PlannerInfo *root, AppendPath *best_path, int flags)
 		Plan	   *subplan;
 
 		/* Must insist that all children return the same tlist */
-		subplan = create_plan_recurse(root, subpath, CP_EXACT_TLIST);
+		subplan = create_append_child_plan(root, (Path *) best_path, subpath);
 
 		/*
 		 * For ordered Appends, we must insert a Sort node if subplan isn't
@@ -1530,7 +1586,7 @@ create_merge_append_plan(PlannerInfo *root, MergeAppendPath *best_path,
 
 		/* Build the child plan */
 		/* Must insist that all children return the same tlist */
-		subplan = create_plan_recurse(root, subpath, CP_EXACT_TLIST);
+		subplan = create_append_child_plan(root, (Path *) best_path, subpath);
 
 		/* Compute sort column info, and adjust subplan's tlist as needed */
 		subplan = prepare_sort_from_pathkeys(subplan, pathkeys,
diff --git a/src/backend/optimizer/plan/planner.c b/src/backend/optimizer/plan/planner.c
index f9a56020f13..a4f688bc03f 100644
--- a/src/backend/optimizer/plan/planner.c
+++ b/src/backend/optimizer/plan/planner.c
@@ -8264,10 +8264,9 @@ apply_scanjoin_target_to_paths(PlannerInfo *root,
 	 * takes from its ORDER BY values, so it can become cheaper relative to
 	 * the others; restore the cost ordering afterwards.
 	 *
-	 * An index path may emit its own target, rel->reltarget plus a value it
-	 * reads from the index (see indexonly_path_target() and
-	 * index_path_orderby_target()).  Such a path needs a projection even when
-	 * the exprs are the same, since its target has more columns than
+	 * A path may emit extra values beyond rel->reltarget (see
+	 * path_extra_exprs()).  Such a path needs a projection even when the
+	 * exprs are the same, since its target has more columns than
 	 * scanjoin_target's sortgrouprefs describe.
 	 */
 	foreach(lc, rel->pathlist)
@@ -8277,7 +8276,7 @@ apply_scanjoin_target_to_paths(PlannerInfo *root,
 		/* Shouldn't have any parameterized paths anymore */
 		Assert(subpath->param_info == NULL);
 
-		if (tlist_same_exprs && subpath->pathtarget == rel->reltarget)
+		if (tlist_same_exprs && !path_emits_extras(subpath))
 			subpath->pathtarget->sortgrouprefs =
 				scanjoin_target->sortgrouprefs;
 		else
@@ -8300,7 +8299,7 @@ apply_scanjoin_target_to_paths(PlannerInfo *root,
 		/* Shouldn't have any parameterized paths anymore */
 		Assert(subpath->param_info == NULL);
 
-		if (tlist_same_exprs && subpath->pathtarget == rel->reltarget)
+		if (tlist_same_exprs && !path_emits_extras(subpath))
 			subpath->pathtarget->sortgrouprefs =
 				scanjoin_target->sortgrouprefs;
 		else
diff --git a/src/backend/optimizer/util/pathnode.c b/src/backend/optimizer/util/pathnode.c
index 6afb5032252..1c79eb62be9 100644
--- a/src/backend/optimizer/util/pathnode.c
+++ b/src/backend/optimizer/util/pathnode.c
@@ -387,24 +387,184 @@ set_cheapest(RelOptInfo *parent_rel)
 }
 
 /*
- * pathtargets_differ
- *	  Do two paths of base relation 'rel' emit different targets?
+ * path_extra_exprs
+ *	  The values a path emits beyond its rel's reltarget.
  *
- * A base relation's paths normally all emit rel->reltarget, but an index
- * path may also emit a value the query needs above the scan, one it has
- * for free and other paths must compute (see indexonly_path_target() and
- * index_path_orderby_target()).  That cost isn't in either path's cost
- * yet; it is charged when the expression is.  So add_path() treats paths
- * with different targets like paths with different pathkeys, and keeps
- * both.
+ * A path normally emits rel->reltarget, the Vars and PlaceHolderVars the
+ * rest of the query needs.  An index path may also emit a non-Var
+ * expression the query needs that it has for free: an index-only scan reads
+ * a returnable index expression column (see indexonly_path_target()), and an
+ * ordering IndexScan has its ORDER BY values (see
+ * index_path_orderby_target()).  A join emits whatever its inputs emit (see
+ * join_path_target()), and so may an Append whose children all emit the
+ * same values (see add_extras_append_path()).  Such a target is a copy of
+ * rel->reltarget with the extra expressions appended, so they're its tail.
+ *
+ * Sort, Material and Memoize don't project and share their input's target,
+ * so they emit what it emits.  Nothing else emits extra values: other path
+ * types build their own targets, and an upper rel's targets differ from its
+ * reltarget for reasons of their own.
+ */
+static ListCell *
+path_extras_start(Path *path, int *nextras)
+{
+	/* see path_extra_exprs() */
+	RelOptInfo *rel = path->parent;
+	PathTarget *reltarget = rel->reltarget;
+	int			nrel;
+
+	*nextras = 0;
+	if (path->pathtarget == reltarget)
+		return NULL;
+
+	switch (nodeTag(path))
+	{
+		case T_SortPath:
+		case T_IncrementalSortPath:
+		case T_MaterialPath:
+		case T_MemoizePath:
+			{
+				Path	   *subpath;
+
+				if (IsA(path, MaterialPath))
+					subpath = ((MaterialPath *) path)->subpath;
+				else if (IsA(path, MemoizePath))
+					subpath = ((MemoizePath *) path)->subpath;
+				else
+					subpath = ((SortPath *) path)->subpath;
+				if (subpath->parent != rel ||
+					subpath->pathtarget != path->pathtarget)
+					return NULL;
+				return path_extras_start(subpath, nextras);
+			}
+		case T_IndexPath:
+			if (!IS_SIMPLE_REL(rel))
+				return NULL;
+			break;
+		case T_NestPath:
+		case T_MergePath:
+		case T_HashPath:
+			if (!IS_JOIN_REL(rel))
+				return NULL;
+			break;
+		case T_AppendPath:
+			/* see add_extras_append_path() */
+			if (rel->reloptkind != RELOPT_BASEREL)
+				return NULL;
+			break;
+		default:
+			return NULL;
+	}
+
+	/*
+	 * Our targets are a copy of reltarget with the values appended.  Anything
+	 * else (say, a final target that apply_projection_to_path() put into an
+	 * index path) isn't ours.
+	 */
+	nrel = list_length(reltarget->exprs);
+	if (list_length(path->pathtarget->exprs) <= nrel)
+		return NULL;
+	for (int i = 0; i < nrel; i++)
+	{
+		if (list_nth(path->pathtarget->exprs, i) !=
+			list_nth(reltarget->exprs, i))
+			return NULL;
+	}
+	*nextras = list_length(path->pathtarget->exprs) - nrel;
+	return list_nth_cell(path->pathtarget->exprs, nrel);
+}
+
+bool
+path_emits_extras(Path *path)
+{
+	int			n;
+
+	return path_extras_start(path, &n) != NULL;
+}
+
+List *
+path_extra_exprs(Path *path)
+{
+	int			n;
+	ListCell   *start = path_extras_start(path, &n);
+
+	if (start == NULL)
+		return NIL;
+	return list_copy_tail(path->pathtarget->exprs,
+						  list_length(path->pathtarget->exprs) - n);
+}
+
+/*
+ * extras_subset
+ *	  Is each of the 'n1' values from 'c1' among the 'n2' from 'c2'?
  */
-static inline bool
-pathtargets_differ(RelOptInfo *rel, Path *path1, Path *path2)
+static bool
+extras_subset(List *l1, ListCell *c1, int n1, List *l2, ListCell *c2, int n2)
 {
-	return rel->reloptkind == RELOPT_BASEREL &&
-		(IsA(path1, IndexPath) || IsA(path2, IndexPath)) &&
-		path1->pathtarget != path2->pathtarget &&
-		!equal(path1->pathtarget->exprs, path2->pathtarget->exprs);
+	int			i1 = list_cell_number(l1, c1);
+	int			i2 = list_cell_number(l2, c2);
+
+	for (int a = 0; a < n1; a++)
+	{
+		Node	   *x = list_nth(l1, i1 + a);
+		bool		found = false;
+
+		for (int b = 0; b < n2 && !found; b++)
+			found = equal(x, list_nth(l2, i2 + b));
+		if (!found)
+			return false;
+	}
+	return true;
+}
+
+/*
+ * compare_path_extras
+ *	  Compare the extra values two paths emit (see path_extra_exprs()).
+ *
+ * A path that emits a superset of what the other emits is "better" in the
+ * same sense as better-sorted: whatever needs the other path's values can
+ * use this one's instead.  The result uses the PathKeysComparison codes so
+ * add_path() can combine it with the pathkeys comparison.
+ */
+static PathKeysComparison
+compare_path_extras(Path *path1, Path *path2)
+{
+	int			n1,
+				n2;
+	ListCell   *c1 = path_extras_start(path1, &n1);
+	ListCell   *c2 = path_extras_start(path2, &n2);
+	List	   *l1 = path1->pathtarget->exprs;
+	List	   *l2 = path2->pathtarget->exprs;
+	bool		sub12,
+				sub21;
+
+	if (n1 == 0 && n2 == 0)
+		return PATHKEYS_EQUAL;
+	sub12 = n1 == 0 || (n2 > 0 && extras_subset(l1, c1, n1, l2, c2, n2));
+	sub21 = n2 == 0 || (n1 > 0 && extras_subset(l2, c2, n2, l1, c1, n1));
+	if (sub12 && sub21)
+		return PATHKEYS_EQUAL;
+	if (sub21)
+		return PATHKEYS_BETTER1;
+	if (sub12)
+		return PATHKEYS_BETTER2;
+	return PATHKEYS_DIFFERENT;
+}
+
+/*
+ * combine_comparisons
+ *	  Combine two PathKeysComparison results into one.
+ *
+ * One path is better overall only if it's better or equal on both counts.
+ */
+static PathKeysComparison
+combine_comparisons(PathKeysComparison a, PathKeysComparison b)
+{
+	if (a == PATHKEYS_EQUAL)
+		return b;
+	if (b == PATHKEYS_EQUAL || a == b)
+		return a;
+	return PATHKEYS_DIFFERENT;
 }
 
 /*
@@ -415,7 +575,11 @@ pathtargets_differ(RelOptInfo *rel, Path *path1, Path *path2)
  *	  A path is worthy if it has a better sort order (better pathkeys) or
  *	  cheaper cost (as defined below), or generates fewer rows, than any
  *    existing path that has the same or superset parameterization rels.  We
- *    also consider parallel-safe paths more worthy than others.
+ *    also consider parallel-safe paths more worthy than others.  And a path
+ *    that emits extra values the query needs (see path_extra_exprs()) is
+ *    better, like a better-sorted one, than a path that emits a subset of
+ *    them: whatever would compute those values above the other path can
+ *    read them from this one.
  *
  *    Cheaper cost can mean either a cheaper total cost or a cheaper startup
  *    cost; if one path is cheaper in one of these aspects and another is
@@ -534,8 +698,8 @@ add_path(RelOptInfo *parent_rel, Path *new_path)
 			old_path_pathkeys = old_path->param_info ? NIL : old_path->pathkeys;
 			keyscmp = compare_pathkeys(new_path_pathkeys,
 									   old_path_pathkeys);
-			if (pathtargets_differ(parent_rel, new_path, old_path))
-				keyscmp = PATHKEYS_DIFFERENT;
+			keyscmp = combine_comparisons(keyscmp,
+										  compare_path_extras(new_path, old_path));
 			if (keyscmp != PATHKEYS_DIFFERENT)
 			{
 				switch (costcmp)
@@ -841,10 +1005,10 @@ add_partial_path(RelOptInfo *parent_rel, Path *new_path)
 		bool		remove_old = false; /* unless new proves superior */
 		PathKeysComparison keyscmp;
 
-		/* Compare pathkeys, and targets as in add_path(). */
+		/* Compare pathkeys, and extra values as in add_path(). */
 		keyscmp = compare_pathkeys(new_path->pathkeys, old_path->pathkeys);
-		if (pathtargets_differ(parent_rel, new_path, old_path))
-			keyscmp = PATHKEYS_DIFFERENT;
+		keyscmp = combine_comparisons(keyscmp,
+									  compare_path_extras(new_path, old_path));
 
 		/*
 		 * Unless pathkeys are incompatible, see if one of the paths dominates
@@ -1113,13 +1277,14 @@ create_samplescan_path(PlannerInfo *root, RelOptInfo *rel, Relids required_outer
  * give the path a copy of rel->reltarget with those expressions appended.
  * They cost the scan nothing (path_target_cost() exempts them), and
  * setrefs.c's fix_join_expr() matches them in the parent join's expressions,
- * replacing its copy with a reference to the scan's output.
+ * replacing its copy with a reference to the scan's output.  A join passes
+ * them on up (see join_path_target()).
  *
- * Restricted to plain base relations: an appendrel child's paths are wrapped
- * by Append/MergeAppend, which build their own targetlists, and the top-level
- * scan/join target already handles the unjoined case.  Returns
- * rel->reltarget unchanged when there's nothing to add, which is the common
- * case, so most paths share it as before.
+ * Restricted to plain base relations: an appendrel child's ORDER BY values
+ * reach the scan/join target through the child's own projection (see
+ * apply_scanjoin_target_to_paths()).  Returns rel->reltarget unchanged when
+ * there's nothing to add, which is the common case, so most paths share it
+ * as before.
  */
 static PathTarget *
 index_path_orderby_target(PlannerInfo *root, IndexOptInfo *index,
@@ -1173,40 +1338,108 @@ index_path_orderby_target(PlannerInfo *root, IndexOptInfo *index,
 	return set_pathtarget_cost_width(root, target);
 }
 
+/*
+ * expr_contains
+ *	  Is 'expr', or some subexpression of it, equal() to 'sub'?
+ */
+static bool
+expr_contains_walker(Node *node, void *context)
+{
+	if (node == NULL)
+		return false;
+	if (equal(node, context))
+		return true;
+	return expression_tree_walker(node, expr_contains_walker, context);
+}
+
+bool
+expr_contains(Node *expr, Node *sub)
+{
+	return expr_contains_walker(expr, sub);
+}
+
+/*
+ * expr_worth_emitting
+ *	  Is 'expr' expensive enough to be worth emitting as an extra value?
+ *
+ * A path that emits an extra value makes add_path() keep it, and every join
+ * path built on it, alongside paths it would otherwise have discarded (see
+ * compare_path_extras()).  That costs planning time at every join level
+ * above.  For a cheap expression, computing it again costs less than that,
+ * so emit only values that cost more than 10 times cpu_operator_cost per
+ * evaluation, the same test make_sort_input_target() uses to decide that an
+ * expression is worth postponing past a sort.
+ */
+bool
+expr_worth_emitting(PlannerInfo *root, Node *expr)
+{
+	QualCost	cost;
+
+	cost_qual_eval_node(&cost, expr, root);
+	return cost.per_tuple > 10 * cpu_operator_cost;
+}
+
+/*
+ * query_tlist_exprs
+ *	  The query's final targetlist expressions, in terms of 'rel'.
+ *
+ * root->processed_tlist is in terms of the parent of an appendrel; translate
+ * it for a child, so it can be compared with the child's index expressions.
+ */
+static List *
+query_tlist_exprs(PlannerInfo *root, RelOptInfo *rel)
+{
+	List	   *exprs = get_tlist_exprs(root->processed_tlist, true);
+
+	if (rel->reloptkind == RELOPT_OTHER_MEMBER_REL)
+		exprs = (List *) adjust_appendrel_attrs_multilevel(root, (Node *) exprs,
+														   rel,
+														   rel->top_parent);
+	return exprs;
+}
+
 /*
  * indexonly_path_target
  *	  Choose the PathTarget for an index-only scan path.
  *
  * An index-only scan reads a returnable index expression column instead of
  * computing it (set_indexonlyscan_references()), and isn't charged for it
- * (indexonly_target_cost()).  But that only helps if the path survives
+ * (path_target_cost()).  But that only helps if the path survives
  * add_path(), and with rel->reltarget, which has only Vars, nothing
  * distinguishes it from a seq scan that will have to compute the expression
  * later; it loses on cost before the expression is charged to anyone.  So
- * if the query's targetlist has a top-level entry equal() to a returnable
- * index expression, give the path a copy of rel->reltarget with that
- * expression appended.  add_path() keeps paths with different targets
- * (see pathtargets_differ()), and once the scan/join target is applied, the
- * paths that must compute the expression pay for it and this one doesn't.
+ * if the query's targetlist contains a returnable index expression, give
+ * the path a copy of rel->reltarget with that expression appended, making
+ * it an extra value the path emits (see path_extra_exprs()).  add_path()
+ * then won't let a path that lacks it dominate this one on cost alone.
+ * The value travels up through joins (see join_path_target()), and
+ * setrefs.c's fix_join_expr() and fix_upper_expr() replace each occurrence,
+ * at any depth, with a reference to it; the paths that must compute it
+ * instead pay for it when the expression is charged.
+ *
+ * An expression is matched with equal(), so a Var of a rel on the nullable
+ * side of an outer join, which carries varnullingrels in the targetlist,
+ * never matches the index's own: the value read below the join isn't the
+ * one the query asks for above it.
  *
- * This is done only when the rel is the query's sole base relation, where
- * the value goes straight to the scan/join target.  Below a join, the join
- * would have to credit it (as join_target_cost() does for ORDER BY values)
- * and the joinrel's add_path() would have to keep it; not attempted yet.
+ * A partial path gets the extra values only if they're parallel-safe.
  */
 static PathTarget *
-indexonly_path_target(PlannerInfo *root, IndexOptInfo *index)
+indexonly_path_target(PlannerInfo *root, IndexOptInfo *index,
+					  bool partial_path)
 {
 	RelOptInfo *rel = index->rel;
 	PathTarget *target = NULL;
+	List	   *tlist_exprs;
 	ListCell   *lt;
 	int			i = 0;
 
-	if (index->indexprs == NIL || rel->reloptkind != RELOPT_BASEREL ||
-		root->processed_tlist == NIL ||
-		!bms_equal(rel->relids, root->all_query_rels))
+	if (index->indexprs == NIL || !IS_SIMPLE_REL(rel) ||
+		root->processed_tlist == NIL)
 		return rel->reltarget;
 
+	tlist_exprs = query_tlist_exprs(root, rel);
+
 	foreach(lt, index->indextlist)
 	{
 		TargetEntry *itle = lfirst_node(TargetEntry, lt);
@@ -1214,14 +1447,18 @@ indexonly_path_target(PlannerInfo *root, IndexOptInfo *index)
 
 		if (!index->canreturn[i++] || IsA(itle->expr, Var))
 			continue;
+		if (partial_path && !is_parallel_safe(root, (Node *) itle->expr))
+			continue;
+		if (!expr_worth_emitting(root, (Node *) itle->expr))
+			continue;
 
-		foreach(lq, root->processed_tlist)
+		foreach(lq, tlist_exprs)
 		{
-			if (equal(lfirst_node(TargetEntry, lq)->expr, itle->expr))
+			if (expr_contains(lfirst(lq), (Node *) itle->expr))
 			{
 				if (target == NULL)
 					target = copy_pathtarget(rel->reltarget);
-				add_column_to_pathtarget(target, itle->expr, 0);
+				add_new_column_to_pathtarget(target, itle->expr);
 				break;
 			}
 		}
@@ -1230,46 +1467,10 @@ indexonly_path_target(PlannerInfo *root, IndexOptInfo *index)
 	if (target == NULL)
 		return rel->reltarget;
 
-	/* add_column_to_pathtarget doesn't maintain cost and width */
+	/* add_new_column_to_pathtarget doesn't maintain cost and width */
 	return set_pathtarget_cost_width(root, target);
 }
 
-/*
- * index_path_emits_orderby
- *	  Does this plain IndexScan path's target hold a value it takes from its
- *	  ORDER BY values?
- *
- * use_physical_tlist() must not replace such a target with the scan's
- * physical tlist, which has only Vars, or whatever needs the value would
- * compute it again.
- */
-bool
-index_path_emits_orderby(IndexPath *ipath)
-{
-	ListCell   *lc;
-
-	if (ipath->path.pathtype != T_IndexScan || ipath->indexorderbys == NIL)
-		return false;
-
-	foreach(lc, ipath->path.pathtarget->exprs)
-	{
-		Expr	   *expr = (Expr *) lfirst(lc);
-		ListCell   *lo,
-				   *lcol;
-
-		if (IsA(expr, Var))
-			continue;
-		forboth(lo, ipath->indexorderbys, lcol, ipath->indexorderbycols)
-		{
-			if (index_orderby_returnable(ipath->indexinfo, lfirst_int(lcol),
-										 (Expr *) lfirst(lo)) &&
-				orderby_tlist_match(expr, (Expr *) lfirst(lo)))
-				return true;
-		}
-	}
-	return false;
-}
-
 /*
  * create_index_path
  *	  Creates a path node for an index scan.
@@ -1310,7 +1511,7 @@ create_index_path(PlannerInfo *root,
 	pathnode->path.pathtype = indexonly ? T_IndexOnlyScan : T_IndexScan;
 	pathnode->path.parent = rel;
 	pathnode->path.pathtarget = indexonly ?
-		indexonly_path_target(root, index) :
+		indexonly_path_target(root, index, partial_path) :
 		index_path_orderby_target(root, index, indexorderbys,
 								  indexorderbycols);
 	pathnode->path.param_info = get_baserel_parampathinfo(root, rel,
@@ -1925,7 +2126,8 @@ create_material_path(RelOptInfo *rel, Path *subpath, bool enabled)
 
 	pathnode->path.pathtype = T_Material;
 	pathnode->path.parent = rel;
-	pathnode->path.pathtarget = rel->reltarget;
+	/* Material doesn't project, so use source path's pathtarget */
+	pathnode->path.pathtarget = subpath->pathtarget;
 	pathnode->path.param_info = subpath->param_info;
 	pathnode->path.parallel_aware = false;
 	pathnode->path.parallel_safe = rel->consider_parallel &&
@@ -1961,7 +2163,8 @@ create_memoize_path(PlannerInfo *root, RelOptInfo *rel, Path *subpath,
 
 	pathnode->path.pathtype = T_Memoize;
 	pathnode->path.parent = rel;
-	pathnode->path.pathtarget = rel->reltarget;
+	/* Memoize doesn't project, so use source path's pathtarget */
+	pathnode->path.pathtarget = subpath->pathtarget;
 	pathnode->path.param_info = subpath->param_info;
 	pathnode->path.parallel_aware = false;
 	pathnode->path.parallel_safe = rel->consider_parallel &&
@@ -2543,6 +2746,80 @@ calc_non_nestloop_required_outer(Path *outer_path, Path *inner_path)
 	return required_outer;
 }
 
+/*
+ * join_path_target
+ *	  Choose the PathTarget for a join path.
+ *
+ * A join normally emits joinrel->reltarget.  If an input emits extra values
+ * (see path_extra_exprs()), the join passes them up too, so that a higher
+ * join or the final scan/join target can use them instead of computing them
+ * (fix_join_expr() matches them in the inputs' targetlists, and
+ * join_free_exprs() credits them).  Without that, an index-only scan or
+ * ordering IndexScan below the topmost join would read the value for
+ * nothing.
+ *
+ * Only values the query still needs above this join are passed up: a
+ * value that refers to a rel on the nullable side of an outer join this
+ * join forms would be wrong above it, but those never match the query's
+ * targetlist (its Vars carry varnullingrels the index's don't), so no input
+ * emits one; and every value we emit came from the final targetlist.
+ *
+ * Values must be parallel-safe if the join is, since a partial join path's
+ * target is computed in workers.
+ *
+ * And only values worth it; see expr_worth_emitting().
+ */
+static PathTarget *
+join_path_target(PlannerInfo *root, RelOptInfo *joinrel,
+				 Path *outer_path, Path *inner_path)
+{
+	List	   *outer_extras = path_extra_exprs(outer_path);
+	List	   *inner_extras = path_extra_exprs(inner_path);
+	PathTarget *target;
+	bool		all_kept = true;
+	ListCell   *lc;
+
+	if (outer_extras == NIL && inner_extras == NIL)
+		return joinrel->reltarget;
+
+	target = copy_pathtarget(joinrel->reltarget);
+	foreach(lc, list_concat(outer_extras, inner_extras))
+	{
+		Node	   *expr = (Node *) lfirst(lc);
+
+		if (expr_worth_emitting(root, expr) &&
+			(!joinrel->consider_parallel || is_parallel_safe(root, expr)))
+			add_new_column_to_pathtarget(target, (Expr *) expr);
+		else
+			all_kept = false;
+	}
+	if (list_length(target->exprs) == list_length(joinrel->reltarget->exprs))
+		return joinrel->reltarget;
+
+	/*
+	 * The values' cost and width are what they added to the inputs' targets,
+	 * since the two inputs' values refer to different rels and can't overlap.
+	 * Working that out is much cheaper than set_pathtarget_cost_width(),
+	 * which looks up each function's cost again.
+	 */
+	if (all_kept)
+	{
+		Path	   *inputs[2] = {outer_path, inner_path};
+
+		for (int i = 0; i < 2; i++)
+		{
+			PathTarget *pt = inputs[i]->pathtarget;
+			PathTarget *rt = inputs[i]->parent->reltarget;
+
+			target->cost.startup += pt->cost.startup - rt->cost.startup;
+			target->cost.per_tuple += pt->cost.per_tuple - rt->cost.per_tuple;
+			target->width += pt->width - rt->width;
+		}
+		return target;
+	}
+	return set_pathtarget_cost_width(root, target);
+}
+
 /*
  * create_nestloop_path
  *	  Creates a pathnode corresponding to a nestloop join between two
@@ -2611,7 +2888,9 @@ create_nestloop_path(PlannerInfo *root,
 
 	pathnode->jpath.path.pathtype = T_NestLoop;
 	pathnode->jpath.path.parent = joinrel;
-	pathnode->jpath.path.pathtarget = joinrel->reltarget;
+	pathnode->jpath.path.pathtarget = join_path_target(root, joinrel,
+													   outer_path,
+													   inner_path);
 	pathnode->jpath.path.param_info =
 		get_joinrel_parampathinfo(root,
 								  joinrel,
@@ -2677,7 +2956,9 @@ create_mergejoin_path(PlannerInfo *root,
 
 	pathnode->jpath.path.pathtype = T_MergeJoin;
 	pathnode->jpath.path.parent = joinrel;
-	pathnode->jpath.path.pathtarget = joinrel->reltarget;
+	pathnode->jpath.path.pathtarget = join_path_target(root, joinrel,
+													   outer_path,
+													   inner_path);
 	pathnode->jpath.path.param_info =
 		get_joinrel_parampathinfo(root,
 								  joinrel,
@@ -2742,7 +3023,9 @@ create_hashjoin_path(PlannerInfo *root,
 
 	pathnode->jpath.path.pathtype = T_HashJoin;
 	pathnode->jpath.path.parent = joinrel;
-	pathnode->jpath.path.pathtarget = joinrel->reltarget;
+	pathnode->jpath.path.pathtarget = join_path_target(root, joinrel,
+													   outer_path,
+													   inner_path);
 	pathnode->jpath.path.param_info =
 		get_joinrel_parampathinfo(root,
 								  joinrel,
@@ -2879,124 +3162,148 @@ orderby_tlist_match(Expr *expr, Expr *orderby)
 }
 
 /*
- * indexonly_target_cost
- *	  path_target_cost() for an index-only scan path.
+ * credit_free_exprs_walker
+ *	  Subtract from context->cost the cost of each outermost subexpression
+ *	  of 'node' equal() to one of context->free.
  *
- * setrefs.c's set_indexonlyscan_references() matches target expressions
- * against the index's returnable columns, expression columns included, and
- * replaces a match with a reference to that column: the scan reads the
- * value the index stored instead of computing it.  So a top-level target
- * expression equal() to a returnable index expression costs nothing.  Only
- * top-level entries are considered, as elsewhere; that's what's normally in
- * a scan's target.
+ * setrefs.c's fix_upper_expr() and fix_join_expr() try to match every
+ * non-Var subexpression against the input's targetlist before descending
+ * into it, and replace a match with a reference to the input's column.  So
+ * an expression costs nothing where an input emits it, wherever it appears
+ * in the expression, and its own subexpressions aren't evaluated either.
  */
-static QualCost
-indexonly_target_cost(PlannerInfo *root, IndexPath *ipath, PathTarget *target)
+typedef struct
 {
-	QualCost	cost = target->cost;
-	IndexOptInfo *index = ipath->indexinfo;
-	ListCell   *lc;
-
-	if (index->indexprs == NIL)
-		return cost;			/* only Vars, which cost nothing anyway */
+	PlannerInfo *root;
+	List	   *free;
+	QualCost	cost;
+}			credit_free_context;
 
-	foreach(lc, target->exprs)
+static bool
+credit_free_exprs_walker(Node *node, credit_free_context * context)
+{
+	if (node == NULL || IsA(node, Var) || IsA(node, Const))
+		return false;
+	if (list_member(context->free, node))
 	{
-		Node	   *expr = (Node *) lfirst(lc);
-		ListCell   *lt;
-		int			i = 0;
-
-		if (IsA(expr, Var))
-			continue;
+		QualCost	ecost;
 
-		foreach(lt, index->indextlist)
-		{
-			TargetEntry *tle = lfirst_node(TargetEntry, lt);
-
-			if (index->canreturn[i++] && !IsA(tle->expr, Var) &&
-				equal(expr, tle->expr))
-			{
-				QualCost	ecost;
-
-				cost_qual_eval_node(&ecost, expr, root);
-				cost.startup -= ecost.startup;
-				cost.per_tuple -= ecost.per_tuple;
-				break;
-			}
-		}
+		cost_qual_eval_node(&ecost, node, context->root);
+		context->cost.startup -= ecost.startup;
+		context->cost.per_tuple -= ecost.per_tuple;
+		return false;			/* don't count its subexpressions twice */
 	}
+	return expression_tree_walker(node, credit_free_exprs_walker, context);
+}
 
-	return cost;
+/*
+ * free_exprs_cost
+ *	  target->cost, less the cost of the values in 'free' that 'target'
+ *	  uses (see credit_free_exprs_walker()).
+ */
+static QualCost
+free_exprs_cost(PlannerInfo *root, PathTarget *target, List *free)
+{
+	credit_free_context context;
+
+	context.root = root;
+	context.free = free;
+	context.cost = target->cost;
+	if (free != NIL)
+		(void) credit_free_exprs_walker((Node *) target->exprs, &context);
+	return context.cost;
 }
 
 /*
- * join_target_cost
- *	  path_target_cost() for a join path.
+ * indexonly_free_exprs
+ *	  The returnable expression columns of an index-only scan's index.
  *
- * A join takes any top-level target expression that one of its inputs
- * already emits from that input's output instead of evaluating it again:
- * setrefs.c's fix_join_expr() matches whole non-Var expressions against the
- * input targetlists before descending into them.  Join inputs ordinarily
- * emit only Vars and PlaceHolderVars, so this changes nothing; but an
- * ordering IndexScan input may emit the ORDER BY values it returns (see
- * index_path_orderby_target()).  Look through Material and Memoize, which
- * pass their input's targetlist up unchanged.  We look only at the join's
- * immediate inputs; the value can't reach a higher join without passing
- * through this one's targetlist, which is the joinrel's and has only Vars.
+ * set_indexonlyscan_references() replaces each occurrence of one of these
+ * in the scan's targetlist and quals with a reference to the index column.
  */
-static QualCost
-join_target_cost(PlannerInfo *root, JoinPath *jpath, PathTarget *target)
+static List *
+indexonly_free_exprs(IndexPath *ipath)
 {
-	QualCost	cost = target->cost;
-	Path	   *inputs[2] = {jpath->outerjoinpath, jpath->innerjoinpath};
-	ListCell   *lc;
+	IndexOptInfo *index = ipath->indexinfo;
+	List	   *result = NIL;
+	ListCell   *lt;
+	int			i = 0;
 
-	foreach(lc, target->exprs)
+	foreach(lt, index->indextlist)
 	{
-		Node	   *expr = (Node *) lfirst(lc);
-
-		if (IsA(expr, Var) || IsA(expr, PlaceHolderVar))
-			continue;
+		TargetEntry *tle = lfirst_node(TargetEntry, lt);
 
-		for (int i = 0; i < 2; i++)
-		{
-			Path	   *input = inputs[i];
+		if (index->canreturn[i++] && !IsA(tle->expr, Var))
+			result = lappend(result, tle->expr);
+	}
+	return result;
+}
 
-			while (IsA(input, MaterialPath) || IsA(input, MemoizePath))
-				input = IsA(input, MaterialPath) ?
-					((MaterialPath *) input)->subpath :
-					((MemoizePath *) input)->subpath;
+/*
+ * indexonly_qual_cost
+ *	  'cost', the cost of an index-only scan's qpquals, less the returnable
+ *	  index expressions they use.
+ *
+ * set_indexonlyscan_references() rewrites the scan's quals as it does its
+ * targetlist.
+ */
+QualCost
+indexonly_qual_cost(PlannerInfo *root, IndexPath *ipath, List *qpquals,
+					QualCost cost)
+{
+	credit_free_context context;
+	ListCell   *lc;
 
-			if (IsA(input, IndexPath) &&
-				index_path_emits_orderby((IndexPath *) input) &&
-				list_member(input->pathtarget->exprs, expr))
-			{
-				QualCost	ecost;
+	context.root = root;
+	context.free = indexonly_free_exprs(ipath);
+	context.cost = cost;
+	if (context.free == NIL)
+		return cost;
+	foreach(lc, qpquals)
+	{
+		Node	   *qual = (Node *) lfirst(lc);
 
-				cost_qual_eval_node(&ecost, expr, root);
-				cost.startup -= ecost.startup;
-				cost.per_tuple -= ecost.per_tuple;
-				break;
-			}
-		}
+		if (IsA(qual, RestrictInfo))
+			qual = (Node *) ((RestrictInfo *) qual)->clause;
+		(void) credit_free_exprs_walker(qual, &context);
 	}
+	return context.cost;
+}
 
-	return cost;
+/*
+ * join_free_exprs
+ *	  The values a join's inputs emit beyond their reltargets.
+ *
+ * fix_join_expr() replaces each occurrence of one of these in the join's
+ * targetlist with a reference to the input's output.  Material and Memoize
+ * pass their input's targetlist up unchanged; path_extra_exprs() already
+ * looks through them.
+ */
+static List *
+join_free_exprs(JoinPath *jpath)
+{
+	return list_concat_unique(path_extra_exprs(jpath->outerjoinpath),
+							  path_extra_exprs(jpath->innerjoinpath));
 }
 
 /*
  * path_target_cost
  *	  Return the cost 'path' pays to evaluate 'target'.
  *
- * Normally that's just target->cost.  A join path doesn't pay for entries
- * an input already emits (see join_target_cost()), nor an index-only scan
- * for entries it reads from the index (see indexonly_target_cost()).  And a
- * plain IndexScan takes any
- * top-level target expression that is a returnable ORDER BY expression
- * (see orderby_tlist_match()) from its ORDER BY values instead of
- * evaluating it (setrefs.c rewrites it into a reference to them), so it
- * doesn't pay for those.  Each target entry
- * is counted at most once, as setrefs.c rewrites it once.
+ * Normally that's just target->cost.  But a path doesn't pay for a value
+ * it has for free:
+ *
+ * - a join, for values one of its inputs emits (join_free_exprs());
+ * - an index-only scan, for returnable index expression columns
+ *   (indexonly_free_exprs());
+ *
+ * in either case wherever the value appears in the target, since setrefs.c
+ * replaces every occurrence (see credit_free_exprs_walker()).  And
+ *
+ * - a plain IndexScan, for a top-level target entry that is one of its
+ *   returnable ORDER BY expressions (see orderby_tlist_match()), which
+ *   setrefs.c's replace_orderby_tlist_refs() takes from its ORDER BY values.
+ *   Only top-level entries are rewritten there, so only those count.
  */
 QualCost
 path_target_cost(PlannerInfo *root, Path *path, PathTarget *target)
@@ -3006,10 +3313,12 @@ path_target_cost(PlannerInfo *root, Path *path, PathTarget *target)
 	ListCell   *lc;
 
 	if (IsA(path, NestPath) || IsA(path, MergePath) || IsA(path, HashPath))
-		return join_target_cost(root, (JoinPath *) path, target);
+		return free_exprs_cost(root, target,
+							   join_free_exprs((JoinPath *) path));
 
 	if (IsA(path, IndexPath) && path->pathtype == T_IndexOnlyScan)
-		return indexonly_target_cost(root, (IndexPath *) path, target);
+		return free_exprs_cost(root, target,
+							   indexonly_free_exprs((IndexPath *) path));
 
 	if (!IsA(path, IndexPath) || path->pathtype != T_IndexScan)
 		return cost;
diff --git a/src/include/nodes/pathnodes.h b/src/include/nodes/pathnodes.h
index e7d447e0889..d376af7adc3 100644
--- a/src/include/nodes/pathnodes.h
+++ b/src/include/nodes/pathnodes.h
@@ -3621,6 +3621,9 @@ typedef struct SemiAntiJoinFactors
  *		RIGHT_ANTI/inner_unique joins)
  * param_source_rels are OK targets for parameterization of result paths
  * pgs_mask is a bitmask of PGS_* constants to limit the join strategy
+ * extras_outer and extras_inner are each input rel's cheapest unparameterized
+ *		path that emits extra values, if that isn't its cheapest path anyway
+ *		(see cheapest_extras_path() in joinpath.c), else NULL
  */
 typedef struct JoinPathExtraData
 {
@@ -3631,6 +3634,8 @@ typedef struct JoinPathExtraData
 	SemiAntiJoinFactors semifactors;
 	Relids		param_source_rels;
 	uint64		pgs_mask;
+	struct Path *extras_outer;
+	struct Path *extras_inner;
 } JoinPathExtraData;
 
 /*
diff --git a/src/include/optimizer/pathnode.h b/src/include/optimizer/pathnode.h
index 8fcfe60aba3..b13792b24b3 100644
--- a/src/include/optimizer/pathnode.h
+++ b/src/include/optimizer/pathnode.h
@@ -231,7 +231,12 @@ extern bool index_orderby_returnable(IndexOptInfo *index, int indexcol,
 extern bool orderby_tlist_match(Expr *expr, Expr *orderby);
 extern QualCost path_target_cost(PlannerInfo *root, Path *path,
 								 PathTarget *target);
-extern bool index_path_emits_orderby(IndexPath *ipath);
+extern List *path_extra_exprs(Path *path);
+extern bool path_emits_extras(Path *path);
+extern bool expr_contains(Node *expr, Node *sub);
+extern bool expr_worth_emitting(PlannerInfo *root, Node *expr);
+extern QualCost indexonly_qual_cost(PlannerInfo *root, IndexPath *ipath,
+									List *qpquals, QualCost cost);
 extern ProjectionPath *create_projection_path(PlannerInfo *root,
 											  RelOptInfo *rel,
 											  Path *subpath,
diff --git a/src/test/regress/expected/gist.out b/src/test/regress/expected/gist.out
index 085391685ea..9ecde5b6205 100644
--- a/src/test/regress/expected/gist.out
+++ b/src/test/regress/expected/gist.out
@@ -754,7 +754,7 @@ select gist_ios_total('select gist_ios_costly(i) from gist_ios_cost order by i')
 
 -- With nothing else to recommend it (no qual, no useful order), an
 -- index-only scan that reads the expression still beats a seq scan that has
--- to compute it: add_path() keeps both, since their targets differ.
+-- to compute it: add_path() keeps the path that emits the value.
 explain (verbose, costs off)
 select gist_ios_costly(i) from gist_ios_cost order by gist_ios_costly(i);
                                          QUERY PLAN                                         
@@ -786,18 +786,158 @@ select gist_ios_costly(i) from gist_ios_cost order by i desc limit 3;
          Output: (gist_ios_costly(i)), i
 (4 rows)
 
--- Below a join it isn't considered.
+-- The expression may be part of a larger one, or appear in a qual.
+explain (verbose, costs off)
+select gist_ios_costly(i) + 1 from gist_ios_cost
+  where gist_ios_costly(i) % 3 = 0;
+                                      QUERY PLAN                                      
+--------------------------------------------------------------------------------------
+ Index Only Scan using gist_ios_cost_i_gist_ios_costly_i_idx on pg_temp.gist_ios_cost
+   Output: ((gist_ios_costly(i)) + 1)
+   Filter: (((gist_ios_costly(gist_ios_cost.i)) % 3) = 0)
+(3 rows)
+
+-- Below a join, the scan emits it and the join passes it up, through more
+-- than one join level.
+create temp table gist_ios_dim (i int, tag text);
+insert into gist_ios_dim select g, 't' || g from generate_series(1, 10000, 7) g;
+create index on gist_ios_dim (i);
+create temp table gist_ios_dim2 (k int, i int);
+insert into gist_ios_dim2 select g, g * 3 from generate_series(1, 3000) g;
+vacuum analyze gist_ios_dim, gist_ios_dim2;
+explain (verbose, costs off)
+select gist_ios_costly(a.i), d.tag from gist_ios_cost a
+  join gist_ios_dim d on d.i = a.i;
+                                          QUERY PLAN                                          
+----------------------------------------------------------------------------------------------
+ Merge Join
+   Output: (gist_ios_costly(a.i)), d.tag
+   Merge Cond: (a.i = d.i)
+   ->  Index Only Scan using gist_ios_cost_i_gist_ios_costly_i_idx on pg_temp.gist_ios_cost a
+         Output: a.i, (gist_ios_costly(a.i))
+   ->  Index Scan using gist_ios_dim_i_idx on pg_temp.gist_ios_dim d
+         Output: d.i, d.tag
+(7 rows)
+
+set enable_hashjoin = off;
+set enable_mergejoin = off;
+explain (verbose, costs off)
+select gist_ios_costly(a.i), d.tag, n.k from gist_ios_cost a
+  join gist_ios_dim d on d.i = a.i join gist_ios_dim2 n on n.i = a.i;
+                                          QUERY PLAN                                          
+----------------------------------------------------------------------------------------------
+ Nested Loop
+   Output: (gist_ios_costly(a.i)), d.tag, n.k
+   Join Filter: (a.i = d.i)
+   ->  Nested Loop
+         Output: d.tag, d.i, n.k, n.i
+         ->  Seq Scan on pg_temp.gist_ios_dim2 n
+               Output: n.k, n.i
+         ->  Index Scan using gist_ios_dim_i_idx on pg_temp.gist_ios_dim d
+               Output: d.i, d.tag
+               Index Cond: (d.i = n.i)
+   ->  Index Only Scan using gist_ios_cost_i_gist_ios_costly_i_idx on pg_temp.gist_ios_cost a
+         Output: a.i, (gist_ios_costly(a.i))
+         Index Cond: (a.i = n.i)
+(13 rows)
+
+reset enable_hashjoin;
+reset enable_mergejoin;
+-- But not from the nullable side of an outer join: there the query's Var
+-- differs from the index's (it may be null), so the value read below the
+-- join isn't the one asked for.
+explain (verbose, costs off)
+select coalesce(gist_ios_costly(a.i), -1) from gist_ios_dim d
+  left join gist_ios_cost a on a.i = d.i * 2;
+                       QUERY PLAN                        
+---------------------------------------------------------
+ Hash Left Join
+   Output: COALESCE(gist_ios_costly(a.i), '-1'::integer)
+   Hash Cond: ((d.i * 2) = a.i)
+   ->  Seq Scan on pg_temp.gist_ios_dim d
+         Output: d.i, d.tag
+   ->  Hash
+         Output: a.i
+         ->  Seq Scan on pg_temp.gist_ios_cost a
+               Output: a.i
+(9 rows)
+
+-- A partitioned table whose partitions all have the index: each child reads
+-- the value and the Append passes it up, even when a partition's columns
+-- are in a different order.
+create temp table gist_ios_part (i int, pad text) partition by range (i);
+create temp table gist_ios_part1 (pad text, i int);
+alter table gist_ios_part attach partition gist_ios_part1
+  for values from (1) to (5001);
+create temp table gist_ios_part2 partition of gist_ios_part
+  for values from (5001) to (10001);
+insert into gist_ios_part select g, 'p' from generate_series(1, 10000) g;
+create index on gist_ios_part (i, gist_ios_costly(i));
+vacuum analyze gist_ios_part;
+explain (verbose, costs off)
+select gist_ios_costly(p.i), d.tag from gist_ios_part p
+  join gist_ios_dim d on d.i = p.i;
+                                               QUERY PLAN                                               
+--------------------------------------------------------------------------------------------------------
+ Hash Join
+   Output: (gist_ios_costly(p.i)), d.tag
+   Hash Cond: (p.i = d.i)
+   ->  Append
+         ->  Index Only Scan using gist_ios_part1_i_gist_ios_costly_i_idx on pg_temp.gist_ios_part1 p_1
+               Output: p_1.i, (gist_ios_costly(p_1.i))
+         ->  Index Only Scan using gist_ios_part2_i_gist_ios_costly_i_idx on pg_temp.gist_ios_part2 p_2
+               Output: p_2.i, (gist_ios_costly(p_2.i))
+   ->  Hash
+         Output: d.tag, d.i
+         ->  Seq Scan on pg_temp.gist_ios_dim d
+               Output: d.tag, d.i
+(12 rows)
+
+-- An index created on only one partition doesn't count: every child must
+-- emit the value for the Append to.
+create temp table gist_ios_part_x (i int) partition by range (i);
+create temp table gist_ios_part_x1 partition of gist_ios_part_x
+  for values from (1) to (5001);
+create temp table gist_ios_part_x2 partition of gist_ios_part_x
+  for values from (5001) to (10001);
+insert into gist_ios_part_x select g from generate_series(1, 10000) g;
+create index on gist_ios_part_x1 (i, gist_ios_costly(i));
+vacuum analyze gist_ios_part_x;
 explain (costs off)
-select gist_ios_costly(a.i) from gist_ios_cost a join gist_ios_cost b using (i);
-               QUERY PLAN                
------------------------------------------
+select gist_ios_costly(p.i), d.tag from gist_ios_part_x p
+  join gist_ios_dim d on d.i = p.i;
+                  QUERY PLAN                  
+----------------------------------------------
  Hash Join
-   Hash Cond: (a.i = b.i)
-   ->  Seq Scan on gist_ios_cost a
+   Hash Cond: (p.i = d.i)
+   ->  Append
+         ->  Seq Scan on gist_ios_part_x1 p_1
+         ->  Seq Scan on gist_ios_part_x2 p_2
    ->  Hash
-         ->  Seq Scan on gist_ios_cost b
-(5 rows)
+         ->  Seq Scan on gist_ios_dim d
+(7 rows)
 
+-- Each of these returns what the same query does without index-only scans.
+create temp table gist_ios_check as
+select gist_ios_costly(a.i) as e, d.tag from gist_ios_cost a
+  join gist_ios_dim d on d.i = a.i;
+set enable_indexonlyscan = off;
+select count(*) as mismatches from (
+  (select gist_ios_costly(a.i) as e, d.tag from gist_ios_cost a
+    join gist_ios_dim d on d.i = a.i
+   except all select * from gist_ios_check)
+  union all
+  (select * from gist_ios_check
+   except all select gist_ios_costly(a.i), d.tag from gist_ios_cost a
+    join gist_ios_dim d on d.i = a.i)) s;
+ mismatches 
+------------
+          0
+(1 row)
+
+reset enable_indexonlyscan;
+drop table gist_ios_check, gist_ios_part, gist_ios_part_x, gist_ios_dim,
+  gist_ios_dim2;
 drop function gist_ios_total(text);
 drop table gist_ios_cost;
 drop function gist_ios_costly(int);
diff --git a/src/test/regress/sql/gist.sql b/src/test/regress/sql/gist.sql
index d032926667b..11d191299dd 100644
--- a/src/test/regress/sql/gist.sql
+++ b/src/test/regress/sql/gist.sql
@@ -375,7 +375,7 @@ select gist_ios_total('select gist_ios_costly(i) from gist_ios_cost order by i')
   < 10000 as ios_expr_is_free;
 -- With nothing else to recommend it (no qual, no useful order), an
 -- index-only scan that reads the expression still beats a seq scan that has
--- to compute it: add_path() keeps both, since their targets differ.
+-- to compute it: add_path() keeps the path that emits the value.
 explain (verbose, costs off)
 select gist_ios_costly(i) from gist_ios_cost order by gist_ios_costly(i);
 explain (verbose, costs off)
@@ -384,9 +384,78 @@ select gist_ios_costly(i) from gist_ios_cost;
 -- still reads the expression from the index.
 explain (verbose, costs off)
 select gist_ios_costly(i) from gist_ios_cost order by i desc limit 3;
--- Below a join it isn't considered.
+-- The expression may be part of a larger one, or appear in a qual.
+explain (verbose, costs off)
+select gist_ios_costly(i) + 1 from gist_ios_cost
+  where gist_ios_costly(i) % 3 = 0;
+-- Below a join, the scan emits it and the join passes it up, through more
+-- than one join level.
+create temp table gist_ios_dim (i int, tag text);
+insert into gist_ios_dim select g, 't' || g from generate_series(1, 10000, 7) g;
+create index on gist_ios_dim (i);
+create temp table gist_ios_dim2 (k int, i int);
+insert into gist_ios_dim2 select g, g * 3 from generate_series(1, 3000) g;
+vacuum analyze gist_ios_dim, gist_ios_dim2;
+explain (verbose, costs off)
+select gist_ios_costly(a.i), d.tag from gist_ios_cost a
+  join gist_ios_dim d on d.i = a.i;
+set enable_hashjoin = off;
+set enable_mergejoin = off;
+explain (verbose, costs off)
+select gist_ios_costly(a.i), d.tag, n.k from gist_ios_cost a
+  join gist_ios_dim d on d.i = a.i join gist_ios_dim2 n on n.i = a.i;
+reset enable_hashjoin;
+reset enable_mergejoin;
+-- But not from the nullable side of an outer join: there the query's Var
+-- differs from the index's (it may be null), so the value read below the
+-- join isn't the one asked for.
+explain (verbose, costs off)
+select coalesce(gist_ios_costly(a.i), -1) from gist_ios_dim d
+  left join gist_ios_cost a on a.i = d.i * 2;
+-- A partitioned table whose partitions all have the index: each child reads
+-- the value and the Append passes it up, even when a partition's columns
+-- are in a different order.
+create temp table gist_ios_part (i int, pad text) partition by range (i);
+create temp table gist_ios_part1 (pad text, i int);
+alter table gist_ios_part attach partition gist_ios_part1
+  for values from (1) to (5001);
+create temp table gist_ios_part2 partition of gist_ios_part
+  for values from (5001) to (10001);
+insert into gist_ios_part select g, 'p' from generate_series(1, 10000) g;
+create index on gist_ios_part (i, gist_ios_costly(i));
+vacuum analyze gist_ios_part;
+explain (verbose, costs off)
+select gist_ios_costly(p.i), d.tag from gist_ios_part p
+  join gist_ios_dim d on d.i = p.i;
+-- An index created on only one partition doesn't count: every child must
+-- emit the value for the Append to.
+create temp table gist_ios_part_x (i int) partition by range (i);
+create temp table gist_ios_part_x1 partition of gist_ios_part_x
+  for values from (1) to (5001);
+create temp table gist_ios_part_x2 partition of gist_ios_part_x
+  for values from (5001) to (10001);
+insert into gist_ios_part_x select g from generate_series(1, 10000) g;
+create index on gist_ios_part_x1 (i, gist_ios_costly(i));
+vacuum analyze gist_ios_part_x;
 explain (costs off)
-select gist_ios_costly(a.i) from gist_ios_cost a join gist_ios_cost b using (i);
+select gist_ios_costly(p.i), d.tag from gist_ios_part_x p
+  join gist_ios_dim d on d.i = p.i;
+-- Each of these returns what the same query does without index-only scans.
+create temp table gist_ios_check as
+select gist_ios_costly(a.i) as e, d.tag from gist_ios_cost a
+  join gist_ios_dim d on d.i = a.i;
+set enable_indexonlyscan = off;
+select count(*) as mismatches from (
+  (select gist_ios_costly(a.i) as e, d.tag from gist_ios_cost a
+    join gist_ios_dim d on d.i = a.i
+   except all select * from gist_ios_check)
+  union all
+  (select * from gist_ios_check
+   except all select gist_ios_costly(a.i), d.tag from gist_ios_cost a
+    join gist_ios_dim d on d.i = a.i)) s;
+reset enable_indexonlyscan;
+drop table gist_ios_check, gist_ios_part, gist_ios_part_x, gist_ios_dim,
+  gist_ios_dim2;
 drop function gist_ios_total(text);
 drop table gist_ios_cost;
 drop function gist_ios_costly(int);
-- 
2.50.1

