From a57358c06115fbddb1631261edb87f43766917da Mon Sep 17 00:00:00 2001 From: "ZizhuanLiu(X-MAN)" <44973863@qq.com> Date: Thu, 8 Oct 2026 16:08:28 +0800 Subject: [PATCH v10] Optimize MCV statistics for sortable types Preserve MCV values in ascending order for sortable data types by introducing a new statistic kind, STATISTIC_KIND_MCV_VALUE_SORTED. This allows planner code to exploit the ordering of MCV values instead of repeatedly comparing every MCV entry with a constant. STATISTIC_KIND_MCV and STATISTIC_KIND_MCV_VALUE_SORTED do not coexist in the same statistics slot. For sortable data types, the sorted MCV list replaces the legacy STATISTIC_KIND_MCV entry, so no additional statistics slot is required. In compute_scalar_stats(), retain the existing MCV generation logic and use an additional ScalarMCVItem workspace to produce the value-sorted MCV list. Use the sorted MCV values during selectivity estimation and range detection. The optimizations are applied conservatively, with strict conditions on the statistics kind, collation, data type, and operator ordering compatibility: * In var_eq_const(), use the minimum and maximum MCV values to quickly determine whether the constant can match an MCV entry, and use binary search when the constant falls within the MCV range. * In mcv_selectivity(), use the sorted MCV values to find the range of entries satisfying <, <=, >, or >=, avoiding per-entry operator evaluation. * In get_stats_slot_range(), use the first and last values of a sorted MCV list directly when its ordering is compatible with the requested sort operator, avoiding the O(N) scan otherwise required to determine the minimum and maximum values. * In eqjoinsel_find_matches(), for the non-hash matching path, use the sorted MCV values of sslot2 to avoid unnecessary comparisons for each MCV value in sslot1. When sslot1 is also STATISTIC_KIND_MCV_VALUE_SORTED, the ordering of both MCV lists can be exploited to further reduce the number of comparisons. When the conditions for these optimizations are not satisfied, retain the existing behavior. The optimization reduces the number of expensive MCV comparisons while preserving the existing behavior for non-sortable types and statistics without a compatible ordering. Update the pg_stats and pg_stats_ext_exprs view to expose STATISTIC_KIND_MCV_VALUE_SORTED through most_common_vals and most_common_freqs. Add an optional most_common_vals_kind parameter to pg_restore_attribute_stats() and the expr arguments of pg_restore_extended_stats() to specify whether most_common_vals[] and most_common_freqs[] are of type 1-STATISTIC_KIND_MCV or 8-STATISTIC_KIND_MCV_VALUE_SORTED. If the parameter is not specified, it defaults to 1-STATISTIC_KIND_MCV, preserving the existing behavior. Discussion: https://www.postgresql.org/message-id/flat/tencent_0489DF3C961BD4D48D32F04F6D6EB4301308@qq.com Fommitfest: https://commitfest.postgresql.org/patch/7302/ --- contrib/postgres_fdw/postgres_fdw.c | 5 +- src/backend/catalog/system_views.sql | 20 + src/backend/commands/analyze.c | 64 +- src/backend/executor/nodeHash.c | 6 +- src/backend/statistics/attribute_stats.c | 36 +- src/backend/statistics/extended_stats_funcs.c | 27 +- src/backend/statistics/stat_utils.c | 83 +- src/backend/utils/adt/like_support.c | 2 +- src/backend/utils/adt/network_selfuncs.c | 26 +- src/backend/utils/adt/selfuncs.c | 1013 +++++++++++++++-- src/backend/utils/cache/lsyscache.c | 22 + src/include/catalog/pg_statistic.h | 6 + src/include/statistics/stat_utils.h | 5 + src/include/statistics/statistics.h | 3 +- src/include/utils/lsyscache.h | 2 + src/include/utils/selfuncs.h | 2 +- src/test/regress/expected/rules.out | 20 + src/test/regress/expected/stats_import.out | 856 ++++++++++++++ src/test/regress/sql/stats_import.sql | 615 ++++++++++ 19 files changed, 2660 insertions(+), 153 deletions(-) diff --git a/contrib/postgres_fdw/postgres_fdw.c b/contrib/postgres_fdw/postgres_fdw.c index 4ca061c..a2c3d37 100644 --- a/contrib/postgres_fdw/postgres_fdw.c +++ b/contrib/postgres_fdw/postgres_fdw.c @@ -377,6 +377,7 @@ enum AttStatsColumns ATTSTATS_RANGE_LENGTH_HISTOGRAM, ATTSTATS_RANGE_EMPTY_FRAC, ATTSTATS_RANGE_BOUNDS_HISTOGRAM, + ATTSTATS_MOST_COMMON_VALS_KIND, ATTSTATS_NUM_FIELDS, }; @@ -6373,6 +6374,8 @@ import_fetched_statistics(Relation relation, get_opt_value(res, row, ATTSTATS_RANGE_EMPTY_FRAC)); set_text_arg(&args[13], get_opt_value(res, row, ATTSTATS_RANGE_BOUNDS_HISTOGRAM)); + set_int32_arg(&args[14], + get_opt_value(res, row, ATTSTATS_MOST_COMMON_VALS_KIND)); /* Try to import the statistics. */ if (!import_attribute_statistics(relation, attnum, false, @@ -6380,7 +6383,7 @@ import_fetched_statistics(Relation relation, &args[3], &args[4], &args[5], &args[6], &args[7], &args[8], &args[9], &args[10], &args[11], - &args[12], &args[13])) + &args[12], &args[13], &args[14])) { ereport(WARNING, errmsg("could not import statistics for foreign table \"%s.%s\" --- attribute statistics import failed for column \"%s\" of this foreign table", diff --git a/src/backend/catalog/system_views.sql b/src/backend/catalog/system_views.sql index 809b9c0..d3600be 100644 --- a/src/backend/catalog/system_views.sql +++ b/src/backend/catalog/system_views.sql @@ -204,6 +204,11 @@ CREATE VIEW pg_stats WITH (security_barrier) AS WHEN stakind3 = 1 THEN stavalues3 WHEN stakind4 = 1 THEN stavalues4 WHEN stakind5 = 1 THEN stavalues5 + WHEN stakind1 = 8 THEN stavalues1 + WHEN stakind2 = 8 THEN stavalues2 + WHEN stakind3 = 8 THEN stavalues3 + WHEN stakind4 = 8 THEN stavalues4 + WHEN stakind5 = 8 THEN stavalues5 END AS most_common_vals, CASE WHEN stakind1 = 1 THEN stanumbers1 @@ -211,6 +216,11 @@ CREATE VIEW pg_stats WITH (security_barrier) AS WHEN stakind3 = 1 THEN stanumbers3 WHEN stakind4 = 1 THEN stanumbers4 WHEN stakind5 = 1 THEN stanumbers5 + WHEN stakind1 = 8 THEN stanumbers1 + WHEN stakind2 = 8 THEN stanumbers2 + WHEN stakind3 = 8 THEN stanumbers3 + WHEN stakind4 = 8 THEN stanumbers4 + WHEN stakind5 = 8 THEN stanumbers5 END AS most_common_freqs, CASE WHEN stakind1 = 2 THEN stavalues1 @@ -332,6 +342,11 @@ CREATE VIEW pg_stats_ext_exprs WITH (security_barrier) AS WHEN (stat.a).stakind3 = 1 THEN (stat.a).stavalues3 WHEN (stat.a).stakind4 = 1 THEN (stat.a).stavalues4 WHEN (stat.a).stakind5 = 1 THEN (stat.a).stavalues5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stavalues5 END) AS most_common_vals, (CASE WHEN (stat.a).stakind1 = 1 THEN (stat.a).stanumbers1 @@ -339,6 +354,11 @@ CREATE VIEW pg_stats_ext_exprs WITH (security_barrier) AS WHEN (stat.a).stakind3 = 1 THEN (stat.a).stanumbers3 WHEN (stat.a).stakind4 = 1 THEN (stat.a).stanumbers4 WHEN (stat.a).stakind5 = 1 THEN (stat.a).stanumbers5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stanumbers5 END) AS most_common_freqs, (CASE WHEN (stat.a).stakind1 = 2 THEN (stat.a).stavalues1 diff --git a/src/backend/commands/analyze.c b/src/backend/commands/analyze.c index 518d752..d3f194b 100644 --- a/src/backend/commands/analyze.c +++ b/src/backend/commands/analyze.c @@ -2503,6 +2503,9 @@ compute_scalar_stats(VacAttrStatsP stats, int values_cnt = 0; int *tupnoLink; ScalarMCVItem *track; + + /* tracks values sorted by compare_scalars() */ + ScalarMCVItem *track_sorted_values; int track_cnt = 0; int num_mcv = stats->attstattarget; int num_bins = stats->attstattarget; @@ -2511,6 +2514,7 @@ compute_scalar_stats(VacAttrStatsP stats, values = palloc_array(ScalarItem, samplerows); tupnoLink = palloc_array(int, samplerows); track = palloc_array(ScalarMCVItem, num_mcv); + track_sorted_values = palloc_array(ScalarMCVItem, num_mcv); memset(&ssup, 0, sizeof(ssup)); ssup.ssup_cxt = CurrentMemoryContext; @@ -2655,6 +2659,8 @@ compute_scalar_stats(VacAttrStatsP stats, } track[j].count = dups_cnt; track[j].first = i + 1 - dups_cnt; + track_sorted_values[track_cnt - 1].count = dups_cnt; + track_sorted_values[track_cnt - 1].first = i + 1 - dups_cnt; } } dups_cnt = 0; @@ -2791,21 +2797,67 @@ compute_scalar_stats(VacAttrStatsP stats, MemoryContext old_context; Datum *mcv_values; float4 *mcv_freqs; + int index; + bool *in_mcv_list; + + /* + * Mark the entries in track_sorted_values[] that belong to the + * MCV list determined by analyze_mcv_list(). The minimum MCV + * count is track[num_mcv - 1].count. First mark entries with + * counts greater than this value, then mark entries with equal + * counts until num_mcv entries have been marked. + * + * There may be more values with the minimum MCV count than the + * number of remaining MCV entries. We therefore select only + * enough of these equally frequent values to reach num_mcv. Since + * entries with the same count are ordered by value in track[], + * this provides a deterministic tie-breaking rule. + */ + in_mcv_list = palloc0_array(bool, track_cnt); + + index = 0; + for (int j = 0; j < track_cnt; j++) + { + if (track_cnt == num_mcv || + track_sorted_values[j].count > track[num_mcv - 1].count) + { + in_mcv_list[j] = true; + index++; + } + } + for (int j = 0; index < num_mcv && j < track_cnt; j++) + { + if (track_sorted_values[j].count == track[num_mcv - 1].count) + { + in_mcv_list[j] = true; + index++; + } + } + Assert(index == num_mcv); /* Must copy the target values into anl_context */ old_context = MemoryContextSwitchTo(stats->anl_context); mcv_values = palloc_array(Datum, num_mcv); mcv_freqs = palloc_array(float4, num_mcv); - for (i = 0; i < num_mcv; i++) + + index = 0; + for (int j = 0; j < track_cnt; j++) { - mcv_values[i] = datumCopy(values[track[i].first].value, - stats->attrtype->typbyval, - stats->attrtype->typlen); - mcv_freqs[i] = (double) track[i].count / (double) samplerows; + if (in_mcv_list[j]) + { + mcv_values[index] = datumCopy(values[track_sorted_values[j].first].value, + stats->attrtype->typbyval, + stats->attrtype->typlen); + mcv_freqs[index] = (double) track_sorted_values[j].count / + (double) samplerows; + index++; + } } + Assert(index == num_mcv); + MemoryContextSwitchTo(old_context); - stats->stakind[slot_idx] = STATISTIC_KIND_MCV; + stats->stakind[slot_idx] = STATISTIC_KIND_MCV_VALUE_SORTED; stats->staop[slot_idx] = mystats->eqopr; stats->stacoll[slot_idx] = stats->attrcollid; stats->stanumbers[slot_idx] = mcv_freqs; diff --git a/src/backend/executor/nodeHash.c b/src/backend/executor/nodeHash.c index 8825bb6..e80e000 100644 --- a/src/backend/executor/nodeHash.c +++ b/src/backend/executor/nodeHash.c @@ -2449,9 +2449,9 @@ ExecHashBuildSkewHash(HashState *hashstate, HashJoinTable hashtable, if (!HeapTupleIsValid(statsTuple)) return; - if (get_attstatsslot(&sslot, statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS)) + if (get_attstatsslot_mcv(&sslot, statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS)) { double frac; int nbuckets; diff --git a/src/backend/statistics/attribute_stats.c b/src/backend/statistics/attribute_stats.c index 25d8a73..3253590 100644 --- a/src/backend/statistics/attribute_stats.c +++ b/src/backend/statistics/attribute_stats.c @@ -55,6 +55,7 @@ enum attribute_stats_argnum RANGE_LENGTH_HISTOGRAM_ARG, RANGE_EMPTY_FRAC_ARG, RANGE_BOUNDS_HISTOGRAM_ARG, + MOST_COMMON_VALS_KIND_ARG, NUM_ATTRIBUTE_STATS_ARGS }; @@ -78,6 +79,7 @@ static struct StatsArgInfo attarginfo[] = [RANGE_LENGTH_HISTOGRAM_ARG] = {"range_length_histogram", TEXTOID}, [RANGE_EMPTY_FRAC_ARG] = {"range_empty_frac", FLOAT4OID}, [RANGE_BOUNDS_HISTOGRAM_ARG] = {"range_bounds_histogram", TEXTOID}, + [MOST_COMMON_VALS_KIND_ARG] = {"most_common_vals_kind", INT2OID}, [NUM_ATTRIBUTE_STATS_ARGS] = {0} }; @@ -407,10 +409,31 @@ attribute_statistics_update_internal(Oid reloid, } else { - statatt_set_slot(values, nulls, replaces, - STATISTIC_KIND_MCV, - eq_opr, atttypcoll, - stanumbers, false, stavalues, false); + int most_common_vals_kind = STATISTIC_KIND_MCV; + + if (!PG_ARGISNULL(MOST_COMMON_VALS_KIND_ARG)) + { + most_common_vals_kind = PG_GETARG_INT16(MOST_COMMON_VALS_KIND_ARG); + + if (most_common_vals_kind != STATISTIC_KIND_MCV && + most_common_vals_kind != STATISTIC_KIND_MCV_VALUE_SORTED) + { + ereport(WARNING, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("could not parse \"%s\": it should be %d or %d", + "most_common_vals_kind", + STATISTIC_KIND_MCV, + STATISTIC_KIND_MCV_VALUE_SORTED))); + result = false; + } + } + + if (most_common_vals_kind == STATISTIC_KIND_MCV || + most_common_vals_kind == STATISTIC_KIND_MCV_VALUE_SORTED) + statatt_set_slot(values, nulls, replaces, + most_common_vals_kind, + eq_opr, atttypcoll, + stanumbers, false, stavalues, false); } } else @@ -732,7 +755,8 @@ import_attribute_statistics(Relation rel, AttrNumber attnum, bool inherited, const NullableDatum *elem_count_histogram, const NullableDatum *range_length_histogram, const NullableDatum *range_empty_frac, - const NullableDatum *range_bounds_histogram) + const NullableDatum *range_bounds_histogram, + const NullableDatum *most_common_vals_kind) { LOCAL_FCINFO(newfcinfo, NUM_ATTRIBUTE_STATS_ARGS); Oid reloid = RelationGetRelid(rel); @@ -752,6 +776,7 @@ import_attribute_statistics(Relation rel, AttrNumber attnum, bool inherited, Assert(range_length_histogram); Assert(range_empty_frac); Assert(range_bounds_histogram); + Assert(most_common_vals_kind); /* annoyingly, get_attname doesn't check attisdropped */ if (attname == NULL || @@ -789,6 +814,7 @@ import_attribute_statistics(Relation rel, AttrNumber attnum, bool inherited, newfcinfo->args[RANGE_LENGTH_HISTOGRAM_ARG] = *range_length_histogram; newfcinfo->args[RANGE_EMPTY_FRAC_ARG] = *range_empty_frac; newfcinfo->args[RANGE_BOUNDS_HISTOGRAM_ARG] = *range_bounds_histogram; + newfcinfo->args[MOST_COMMON_VALS_KIND_ARG] = *most_common_vals_kind; return attribute_statistics_update_internal(reloid, attname, attnum, inherited, newfcinfo); diff --git a/src/backend/statistics/extended_stats_funcs.c b/src/backend/statistics/extended_stats_funcs.c index 00b8dbe..0679504 100644 --- a/src/backend/statistics/extended_stats_funcs.c +++ b/src/backend/statistics/extended_stats_funcs.c @@ -98,6 +98,7 @@ enum extended_stats_exprs_element RANGE_LENGTH_HISTOGRAM_ELEM, RANGE_EMPTY_FRAC_ELEM, RANGE_BOUNDS_HISTOGRAM_ELEM, + MOST_COMMON_VALS_KIND_ELEM, NUM_ATTRIBUTE_STATS_ELEMS }; @@ -118,7 +119,8 @@ static const char *extexprargname[NUM_ATTRIBUTE_STATS_ELEMS] = "elem_count_histogram", "range_length_histogram", "range_empty_frac", - "range_bounds_histogram" + "range_bounds_histogram", + "most_common_vals_kind" }; static bool extended_statistics_update(FunctionCallInfo fcinfo); @@ -1356,6 +1358,7 @@ import_pg_statistic(Relation pgsd, JsonbContainer *cont, ArrayType *nums_arr = DatumGetArrayTypeP(stanumbers); int nvals = ARR_DIMS(vals_arr)[0]; int nnums = ARR_DIMS(nums_arr)[0]; + int most_common_vals_kind = STATISTIC_KIND_MCV; if (nvals != nnums) { @@ -1367,8 +1370,28 @@ import_pg_statistic(Relation pgsd, JsonbContainer *cont, goto pg_statistic_error; } + if (found[MOST_COMMON_VALS_KIND_ELEM]) + { + s = jbv_string_get_cstr(&val[MOST_COMMON_VALS_KIND_ELEM]); + most_common_vals_kind = atoi(s); + + pfree(s); + } + + if (most_common_vals_kind != STATISTIC_KIND_MCV && + most_common_vals_kind != STATISTIC_KIND_MCV_VALUE_SORTED) + { + ereport(WARNING, + (errcode(ERRCODE_INVALID_PARAMETER_VALUE), + errmsg("could not parse \"%s\": it should be %d or %d", + "most_common_vals_kind", + STATISTIC_KIND_MCV, + STATISTIC_KIND_MCV_VALUE_SORTED))); + goto pg_statistic_error; + } + statatt_set_slot(values, nulls, replaces, - STATISTIC_KIND_MCV, + most_common_vals_kind, basetypcache->eq_opr, typcoll, stanumbers, false, stavalues, false); } diff --git a/src/backend/statistics/stat_utils.c b/src/backend/statistics/stat_utils.c index e8117ed..8403c27 100644 --- a/src/backend/statistics/stat_utils.c +++ b/src/backend/statistics/stat_utils.c @@ -684,7 +684,12 @@ statatt_set_slot(Datum *values, bool *nulls, bool *replaces, if (first_empty < 0 && DatumGetInt16(values[stakind_attnum]) == 0) first_empty = slotidx; - if (DatumGetInt16(values[stakind_attnum]) == stakind) + + if (DatumGetInt16(values[stakind_attnum]) == stakind || + (DatumGetInt16(values[stakind_attnum]) == STATISTIC_KIND_MCV && + stakind == STATISTIC_KIND_MCV_VALUE_SORTED) || + (DatumGetInt16(values[stakind_attnum]) == STATISTIC_KIND_MCV_VALUE_SORTED && + stakind == STATISTIC_KIND_MCV)) break; } @@ -852,3 +857,79 @@ statatt_check_bounds_histogram(Datum arrayval) return true; } + +/* + * get_max_mcv_frequency + * Return the maximum frequency in an MCV statistics slot. + * + * For STATISTIC_KIND_MCV, the MCV entries are sorted by frequency, + * so the maximum frequency is the first entry. + * + * For STATISTIC_KIND_MCV_VALUE_SORTED, the entries are sorted by value, + * so find the maximum frequency by scanning numbers[]. + * + * Return false if the statistics slot is not an MCV slot or contains + * no frequency values. + */ +bool +get_max_mcv_frequency(AttStatsSlot *sslot, int statskind, + double *max_frequency) +{ + int i = 0; + + if ((statskind != STATISTIC_KIND_MCV && + statskind != STATISTIC_KIND_MCV_VALUE_SORTED) || + sslot->nnumbers == 0) + return false; + + if (statskind == STATISTIC_KIND_MCV_VALUE_SORTED) + { + for (int j = 1; j < sslot->nnumbers; j++) + { + if (sslot->numbers[j] > sslot->numbers[i]) + i = j; + } + } + + *max_frequency = sslot->numbers[i]; + + return true; +} + +/* + * get_min_mcv_frequency + * Return the minimum frequency in an MCV statistics slot. + * + * For STATISTIC_KIND_MCV, the MCV entries are sorted by frequency, + * so the minimum frequency is the last entry. + * + * For STATISTIC_KIND_MCV_VALUE_SORTED, the entries are sorted by value, + * so find the minimum frequency by scanning numbers[]. + * + * Return false if the statistics slot is not an MCV slot or contains + * no frequency values. + */ +bool +get_min_mcv_frequency(AttStatsSlot *sslot, int statskind, + double *min_frequency) +{ + int i = sslot->nnumbers - 1; + + if ((statskind != STATISTIC_KIND_MCV && + statskind != STATISTIC_KIND_MCV_VALUE_SORTED) || + sslot->nnumbers == 0) + return false; + + if (statskind == STATISTIC_KIND_MCV_VALUE_SORTED) + { + for (int j = 0; j < sslot->nnumbers - 1; j++) + { + if (sslot->numbers[j] < sslot->numbers[i]) + i = j; + } + } + + *min_frequency = sslot->numbers[i]; + + return true; +} diff --git a/src/backend/utils/adt/like_support.c b/src/backend/utils/adt/like_support.c index 588425b..247d3f4 100644 --- a/src/backend/utils/adt/like_support.c +++ b/src/backend/utils/adt/like_support.c @@ -730,7 +730,7 @@ patternsel_common(PlannerInfo *root, */ mcv_selec = mcv_selectivity(&vardata, &opproc, collation, constval, true, - &sumcommon); + &sumcommon, oprid); /* * Now merge the results from the MCV and histogram calculations, diff --git a/src/backend/utils/adt/network_selfuncs.c b/src/backend/utils/adt/network_selfuncs.c index 2a8d2de..cf15cff 100644 --- a/src/backend/utils/adt/network_selfuncs.c +++ b/src/backend/utils/adt/network_selfuncs.c @@ -148,7 +148,7 @@ networksel(PG_FUNCTION_ARGS) fmgr_info(get_opcode(operator), &proc); mcv_selec = mcv_selectivity(&vardata, &proc, InvalidOid, constvalue, varonleft, - &sumcommon); + &sumcommon, operator); /* * If we have a histogram, use it to estimate the proportion of the @@ -305,9 +305,9 @@ networkjoinsel_inner(Oid operator, int opr_codenum, stats = (Form_pg_statistic) GETSTRUCT(vardata1->statsTuple); nullfrac1 = stats->stanullfrac; - mcv1_exists = get_attstatsslot(&mcv1_slot, vardata1->statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); + mcv1_exists = get_attstatsslot_mcv(&mcv1_slot, vardata1->statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); hist1_exists = get_attstatsslot(&hist1_slot, vardata1->statsTuple, STATISTIC_KIND_HISTOGRAM, InvalidOid, ATTSTATSSLOT_VALUES); @@ -327,9 +327,9 @@ networkjoinsel_inner(Oid operator, int opr_codenum, stats = (Form_pg_statistic) GETSTRUCT(vardata2->statsTuple); nullfrac2 = stats->stanullfrac; - mcv2_exists = get_attstatsslot(&mcv2_slot, vardata2->statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); + mcv2_exists = get_attstatsslot_mcv(&mcv2_slot, vardata2->statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); hist2_exists = get_attstatsslot(&hist2_slot, vardata2->statsTuple, STATISTIC_KIND_HISTOGRAM, InvalidOid, ATTSTATSSLOT_VALUES); @@ -432,9 +432,9 @@ networkjoinsel_semi(Oid operator, int opr_codenum, stats = (Form_pg_statistic) GETSTRUCT(vardata1->statsTuple); nullfrac1 = stats->stanullfrac; - mcv1_exists = get_attstatsslot(&mcv1_slot, vardata1->statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); + mcv1_exists = get_attstatsslot_mcv(&mcv1_slot, vardata1->statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); hist1_exists = get_attstatsslot(&hist1_slot, vardata1->statsTuple, STATISTIC_KIND_HISTOGRAM, InvalidOid, ATTSTATSSLOT_VALUES); @@ -454,9 +454,9 @@ networkjoinsel_semi(Oid operator, int opr_codenum, stats = (Form_pg_statistic) GETSTRUCT(vardata2->statsTuple); nullfrac2 = stats->stanullfrac; - mcv2_exists = get_attstatsslot(&mcv2_slot, vardata2->statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); + mcv2_exists = get_attstatsslot_mcv(&mcv2_slot, vardata2->statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); hist2_exists = get_attstatsslot(&hist2_slot, vardata2->statsTuple, STATISTIC_KIND_HISTOGRAM, InvalidOid, ATTSTATSSLOT_VALUES); diff --git a/src/backend/utils/adt/selfuncs.c b/src/backend/utils/adt/selfuncs.c index 5ee3775..461b4e1 100644 --- a/src/backend/utils/adt/selfuncs.c +++ b/src/backend/utils/adt/selfuncs.c @@ -117,9 +117,11 @@ #include "optimizer/paths.h" #include "optimizer/plancat.h" #include "parser/parse_clause.h" +#include "parser/parse_oper.h" #include "parser/parse_relation.h" #include "parser/parsetree.h" #include "rewrite/rewriteManip.h" +#include "statistics/stat_utils.h" #include "statistics/statistics.h" #include "utils/acl.h" #include "utils/array.h" @@ -135,6 +137,7 @@ #include "utils/selfuncs.h" #include "utils/snapmgr.h" #include "utils/spccache.h" +#include "utils/sortsupport.h" #include "utils/syscache.h" #include "utils/timestamp.h" #include "utils/typcache.h" @@ -154,6 +157,28 @@ #define EQJOINSEL_MCV_HASH_THRESHOLD 20 #endif +/* + * Status codes for IN_MCV_RANGE: + * 0 - Must compare against every MCV entry. + * 1 - Within MCV range; still need to check existence + * via binary search or other means. + * 2 - Outside MCV range; no per-MCV comparison required. + */ +#define IN_MCV_RANGE_UNKNOWN 0 +#define IN_MCV_RANGE_YES 1 +#define IN_MCV_RANGE_NO 2 + +/* Threshold: MCV entries worthy of special comparison (e.g. binary search) */ +#define MCV_SPECIAL_COMPARE_THRESHOLD 3 + +/* + * When using "<" in ApplySortComparator(valueA, x, valueB, y, z), + * indicates valueA's relative position compared with valueB. + */ +#define ON_LEFT(i) ((i) < 0) +#define ON_EQUAL(i) ((i) == 0) +#define ON_RIGHT(i) ((i) > 0) + /* Entries in the simplehash hash table used by eqjoinsel_find_matches */ typedef struct MCVHashEntry { @@ -191,7 +216,8 @@ static double eqjoinsel_inner(FmgrInfo *eqproc, Oid collation, Form_pg_statistic stats1, Form_pg_statistic stats2, bool have_mcvs1, bool have_mcvs2, bool *hasmatch1, bool *hasmatch2, - int *p_nmatches); + int *p_nmatches, + Oid oproid, int statskind1, int statskind2); static double eqjoinsel_semi(FmgrInfo *eqproc, Oid collation, Oid hashLeft, Oid hashRight, bool op_is_reversed, @@ -203,14 +229,16 @@ static double eqjoinsel_semi(FmgrInfo *eqproc, Oid collation, bool have_mcvs1, bool have_mcvs2, bool *hasmatch1, bool *hasmatch2, int *p_nmatches, - RelOptInfo *inner_rel); + RelOptInfo *inner_rel, + Oid eqoproid, int statskind1, int statskind2); static void eqjoinsel_find_matches(FmgrInfo *eqproc, Oid collation, Oid hashLeft, Oid hashRight, bool op_is_reversed, AttStatsSlot *sslot1, AttStatsSlot *sslot2, int nvalues1, int nvalues2, bool *hasmatch1, bool *hasmatch2, - int *p_nmatches, double *p_matchprodfreq); + int *p_nmatches, double *p_matchprodfreq, + Oid eqoproid, int statskind1, int statskind2); static uint32 hash_mcv(MCVHashTable_hash *tab, Datum key); static bool mcvs_equal(MCVHashTable_hash *tab, Datum key0, Datum key1); static bool estimate_multivariate_ndistinct(PlannerInfo *root, @@ -255,7 +283,8 @@ static bool get_variable_range(PlannerInfo *root, VariableStatData *vardata, static void get_stats_slot_range(AttStatsSlot *sslot, Oid opfuncoid, FmgrInfo *opproc, Oid collation, int16 typLen, bool typByVal, - Datum *min, Datum *max, bool *p_have_data); + Datum *min, Datum *max, bool *p_have_data, + int statskind, Oid sortop); static bool get_actual_variable_range(PlannerInfo *root, VariableStatData *vardata, Oid sortop, Oid collation, @@ -411,6 +440,12 @@ var_eq_const(VariableStatData *vardata, Oid oproid, Oid collation, AttStatsSlot sslot; bool match = false; int i; + double sumcommon = 0.0; + int statskind; + + statskind = get_attstatsslot_mcv(&sslot, vardata->statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); /* * Is the constant "=" to any of the column's most common values? @@ -419,12 +454,13 @@ var_eq_const(VariableStatData *vardata, Oid oproid, Oid collation, * don't like this, maybe you shouldn't be using eqsel for your * operator...) */ - if (get_attstatsslot(&sslot, vardata->statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS)) + if (statskind) { LOCAL_FCINFO(fcinfo, 2); FmgrInfo eqproc; + bool scan_entire_mcv = false; + int in_mcv_range = IN_MCV_RANGE_UNKNOWN; + int arg_var; /* Index of the Var in fcinfo->args[] */ fmgr_info(opfuncoid, &eqproc); @@ -440,26 +476,203 @@ var_eq_const(VariableStatData *vardata, Oid oproid, Oid collation, fcinfo->args[1].isnull = false; /* be careful to apply operator right way 'round */ if (varonleft) + { fcinfo->args[1].value = constval; + arg_var = 0; + } else + { fcinfo->args[0].value = constval; + arg_var = 1; + } - for (i = 0; i < sslot.nvalues; i++) + selec = 0.0; + i = 0; + + if (sslot.stacoll != collation && OidIsValid(collation)) { - Datum fresult; + /* + * Scanning the entire MCV array is needed when the collation + * used for the comparison is nondeterministic and differs + * from the statistics collation. In this case, the comparison + * may match multiple MCV values, so we must continue scanning + * after finding a match. + */ + pg_locale_t mylocale = pg_newlocale_from_collation(collation); - if (varonleft) - fcinfo->args[0].value = sslot.values[i]; - else - fcinfo->args[1].value = sslot.values[i]; - fcinfo->isnull = false; - fresult = FunctionCallInvoke(fcinfo); - if (!fcinfo->isnull && DatumGetBool(fresult)) + scan_entire_mcv = !mylocale->deterministic; + } + + if (sslot.stacoll == collation && + statskind == STATISTIC_KIND_MCV_VALUE_SORTED && + sslot.nvalues > MCV_SPECIAL_COMPARE_THRESHOLD && + comparison_ops_are_compatible(sslot.staop, oproid)) + { + /* + * If the statistics and comparison collations match, the MCV + * values are sorted, the operator is compatible with the + * ordering used to sort the MCV values, and the MCV list is + * large enough, optimize the lookup using the sorted MCV + * values: + * + * - If the constant is equal to the first or last MCV value, + * the lookup can be completed immediately. + * + * - If the constant falls within the MCV range, use binary + * search to find a matching value, reducing the number of + * comparisons from O(N) to O(log N). + * + * - If the constant falls outside the MCV range, no MCV value + * can match. Subsequent processing can therefore sum + * sumcommon directly without comparing the constant against + * the MCV values. + */ + Oid ltopr; + + /* Look for default "<" operators for valuetype. */ + get_sort_group_operators(sslot.valuetype, + false, false, false, + <opr, NULL, NULL, + NULL); + if (OidIsValid(ltopr)) { - match = true; - break; + SortSupportData ssup = {0}; + int compare; + + /* And no full MCV scan required */ + scan_entire_mcv = false; + + ssup.ssup_cxt = CurrentMemoryContext; + ssup.ssup_collation = sslot.stacoll; + ssup.ssup_nulls_first = false; + ssup.abbreviate = false; + PrepareSortSupportFromOrderingOp(ltopr, &ssup); + + /* First compare against values[0] */ + compare = ApplySortComparator(constval, false, + sslot.values[0], false, &ssup); + if (ON_EQUAL(compare)) + { + /* + * Constant is "=" to this common value. We know + * selectivity exactly (or as exactly as ANALYZE could + * calculate it, anyway). + */ + match = true; + selec = sslot.numbers[0]; + } + else if (ON_LEFT(compare) || sslot.nvalues == 1) + { + in_mcv_range = IN_MCV_RANGE_NO; + } + else + { + /* Next compare against values[sslot.nvalues - 1] */ + compare = ApplySortComparator(constval, false, + sslot.values[sslot.nvalues - 1], false, &ssup); + if (ON_EQUAL(compare)) + { + /* + * Constant is "=" to this common value. We know + * selectivity exactly (or as exactly as ANALYZE + * could calculate it, anyway). + */ + match = true; + selec = sslot.numbers[sslot.nvalues - 1]; + } + else if (ON_RIGHT(compare)) + in_mcv_range = IN_MCV_RANGE_NO; + else if (ON_LEFT(compare)) + { + int tmp_l_bound; + int tmp_r_bound; + + /* + * For binary search: refers to the first and last + * uncompared elements. + */ + tmp_l_bound = 1; + tmp_r_bound = sslot.nvalues - 2; + + while (tmp_l_bound <= tmp_r_bound) + { + int mid = (tmp_l_bound + tmp_r_bound) / 2; + + compare = ApplySortComparator(constval, false, + sslot.values[mid], false, &ssup); + if (ON_EQUAL(compare)) + { + /* + * Constant is "=" to this common value. + * We know selectivity exactly (or as + * exactly as ANALYZE could calculate it, + * anyway). + */ + match = true; + selec = sslot.numbers[mid]; + + break; + } + else if (ON_RIGHT(compare)) + tmp_l_bound = mid + 1; + else if (ON_LEFT(compare)) + tmp_r_bound = mid - 1; + } + + if (!match) + in_mcv_range = IN_MCV_RANGE_NO; + } + } } } + + if (!match) + { + /* + * Compare the constant expression with the MCVs. + * + * If the constant matches the current MCV, stop here when a + * full MCV scan is not required. Otherwise, continue scanning + * the remaining MCVs and accumulate the selectivity of all + * matching MCVs. + * + * While scanning, also accumulate the selectivity of the MCVs + * examined so far. This is used later to estimate the + * selectivity of a non-NULL constant that does not match any + * MCV. + */ + for (i = 0; i < sslot.nvalues; i++) + { + if (in_mcv_range == IN_MCV_RANGE_UNKNOWN || scan_entire_mcv) + { + Datum fresult; + + fcinfo->args[arg_var].value = sslot.values[i]; + fcinfo->isnull = false; + fresult = FunctionCallInvoke(fcinfo); + if (!fcinfo->isnull && DatumGetBool(fresult)) + { + /* + * Constant is "=" to this common value. We know + * selectivity exactly (or as exactly as ANALYZE + * could calculate it, anyway). + */ + match = true; + if (!scan_entire_mcv) + { + selec = sslot.numbers[i]; + break; + } + + selec += sslot.numbers[i]; + } + } + + sumcommon += sslot.numbers[i]; + } + } + + free_attstatsslot(&sslot); } else { @@ -467,26 +680,16 @@ var_eq_const(VariableStatData *vardata, Oid oproid, Oid collation, i = 0; /* keep compiler quiet */ } - if (match) - { - /* - * Constant is "=" to this common value. We know selectivity - * exactly (or as exactly as ANALYZE could calculate it, anyway). - */ - selec = sslot.numbers[i]; - } - else + if (!match) { /* * Comparison is against a constant that is neither NULL nor any * of the common values. Its selectivity cannot be more than * this: */ - double sumcommon = 0.0; double otherdistinct; + double min_frequency; - for (i = 0; i < sslot.nnumbers; i++) - sumcommon += sslot.numbers[i]; selec = 1.0 - sumcommon - nullfrac; CLAMP_PROBABILITY(selec); @@ -504,11 +707,10 @@ var_eq_const(VariableStatData *vardata, Oid oproid, Oid collation, * Another cross-check: selectivity shouldn't be estimated as more * than the least common "most common value". */ - if (sslot.nnumbers > 0 && selec > sslot.numbers[sslot.nnumbers - 1]) - selec = sslot.numbers[sslot.nnumbers - 1]; + if (get_min_mcv_frequency(&sslot, statskind, &min_frequency) && + selec > min_frequency) + selec = min_frequency; } - - free_attstatsslot(&sslot); } else { @@ -570,6 +772,7 @@ var_eq_non_const(VariableStatData *vardata, Oid oproid, Oid collation, { double ndistinct; AttStatsSlot sslot; + int statskind; /* * Search is for a value that we do not know a priori, but we will @@ -590,12 +793,16 @@ var_eq_non_const(VariableStatData *vardata, Oid oproid, Oid collation, * Cross-check: selectivity should never be estimated as more than the * most common value's. */ - if (get_attstatsslot(&sslot, vardata->statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_NUMBERS)) + statskind = get_attstatsslot_mcv(&sslot, vardata->statsTuple, + InvalidOid, + ATTSTATSSLOT_NUMBERS); + if (statskind) { - if (sslot.nnumbers > 0 && selec > sslot.numbers[0]) - selec = sslot.numbers[0]; + double max_frequency; + + if (get_max_mcv_frequency(&sslot, statskind, &max_frequency) && + selec > max_frequency) + selec = max_frequency; free_attstatsslot(&sslot); } } @@ -753,7 +960,7 @@ scalarineqsel(PlannerInfo *root, Oid operator, bool isgt, bool iseq, * by MCV entries. */ mcv_selec = mcv_selectivity(vardata, &opproc, collation, constval, true, - &sumcommon); + &sumcommon, operator); /* * If there is a histogram, determine which bin the constant falls in, and @@ -805,23 +1012,29 @@ scalarineqsel(PlannerInfo *root, Oid operator, bool isgt, bool iseq, double mcv_selectivity(VariableStatData *vardata, FmgrInfo *opproc, Oid collation, Datum constval, bool varonleft, - double *sumcommonp) + double *sumcommonp, Oid operator) { double mcv_selec, sumcommon; AttStatsSlot sslot; int i; + int statskind; mcv_selec = 0.0; sumcommon = 0.0; if (HeapTupleIsValid(vardata->statsTuple) && statistic_proc_security_check(vardata, opproc->fn_oid) && - get_attstatsslot(&sslot, vardata->statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS)) + (statskind = get_attstatsslot_mcv(&sslot, vardata->statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS))) { LOCAL_FCINFO(fcinfo, 2); + Datum fresult; + int in_mcv_range = IN_MCV_RANGE_UNKNOWN; + int lBound = -1; /* -1 means unknown */ + int rBound = -1; /* -1 means unknown */ + int arg_var; /* Index of the Var in fcinfo->args[] */ /* * We invoke the opproc "by hand" so that we won't fail on NULL @@ -836,24 +1049,408 @@ mcv_selectivity(VariableStatData *vardata, FmgrInfo *opproc, Oid collation, fcinfo->args[1].isnull = false; /* be careful to apply operator right way 'round */ if (varonleft) + { fcinfo->args[1].value = constval; + arg_var = 0; + } else + { fcinfo->args[0].value = constval; + arg_var = 1; + } + + if (sslot.stacoll == collation && + statskind == STATISTIC_KIND_MCV_VALUE_SORTED && + sslot.nvalues > MCV_SPECIAL_COMPARE_THRESHOLD && + comparison_ops_are_compatible(sslot.staop, operator)) + { + /* + * If the statistics and comparison collations match, the MCV + * values are sorted, the comparison operators are compatible, and + * the MCV list is large enough, use the sorted MCV values to find + * the range of entries satisfying the predicate. + */ + Oid ltopr; + + /* Look for default "<" operators for valuetype. */ + get_sort_group_operators(sslot.valuetype, + false, false, false, + <opr, NULL, NULL, + NULL); + if (OidIsValid(ltopr)) + { + SortSupportData ssup = {0}; + Oid opfamily; + Oid opcintype; + CompareType cmptype = COMPARE_INVALID; + int compare; + + ssup.ssup_cxt = CurrentMemoryContext; + ssup.ssup_collation = sslot.stacoll; + ssup.ssup_nulls_first = false; + ssup.abbreviate = false; + PrepareSortSupportFromOrderingOp(ltopr, &ssup); + + if (get_ordering_op_properties(operator, &opfamily, &opcintype, + &cmptype)) + { + if (!varonleft) + { + /* + * The constant is on the left, so reverse the + * operator selectivity function. + */ + if (cmptype == COMPARE_LT || cmptype == COMPARE_GT) + cmptype = cmptype == COMPARE_LT ? COMPARE_GT : COMPARE_LT; + + if (cmptype == COMPARE_LE || cmptype == COMPARE_GE) + cmptype = cmptype == COMPARE_LE ? COMPARE_GE : COMPARE_LE; + } + + if (cmptype == COMPARE_LT || cmptype == COMPARE_LE) + { + /* For predicate: (VAR "<" CONST) or (VAR "<=" CONST) */ + + /* Start by comparing with the Min VAR: values[0] */ + compare = ApplySortComparator(constval, false, + sslot.values[0], false, &ssup); + if (ON_LEFT(compare)) + { + /* + * CONST < Min VAR: all MCV entries > CONST, no + * MCV entries satisfy the predicate. + */ + in_mcv_range = IN_MCV_RANGE_NO; + } + else if (ON_EQUAL(compare)) + { + /* CONST == Min VAR */ + if (cmptype == COMPARE_LT) + { + /* + * When cmptype is "<": but all MCV entries > + * CONST, so no MCV entries satisfy the + * predicate. + */ + in_mcv_range = IN_MCV_RANGE_NO; + } + else if (cmptype == COMPARE_LE) + { + /* + * When cmptype is "<=": only the Min VAR + * satisfy the predicate. + */ + in_mcv_range = IN_MCV_RANGE_YES; + lBound = 0; + rBound = lBound; + } + } + else if (ON_RIGHT(compare)) + { + /* + * Min VAR < CONST. Left boundary determined, now + * find right boundary. + */ + in_mcv_range = IN_MCV_RANGE_YES; + lBound = 0; + + /* compare at Max VAR: values[sslot.nvalues - 1] */ + compare = ApplySortComparator(constval, false, + sslot.values[sslot.nvalues - 1], false, &ssup); + if (ON_RIGHT(compare)) + { + /* + * Max VAR < CONST: all MCV entries "<"/"<=" + * CONST, all MCV entries satisfy the + * predicate. + */ + rBound = sslot.nvalues - 1; + } + else if (ON_EQUAL(compare)) + { + /* Max VAR == CONST */ + if (cmptype == COMPARE_LT) + { + /* + * when cmptype is "<": + * values[sslot.nvalues - 2] is the + * rightmost entry satisfying the + * predicate. + */ + rBound = sslot.nvalues - 2; + } + else if (cmptype == COMPARE_LE) + { + /* + * When cmptype is "<=": all MCV entries + * "<=" CONST, all MCV entries satisfy the + * predicate. + */ + rBound = sslot.nvalues - 1; + } + } + else if (ON_LEFT(compare)) + { + /* + * Max VAR > CONST: binary search. + */ + int tmp_l_bound; + int tmp_r_bound; + + /* + * Initially set right boundary equal to left + * boundary + */ + rBound = lBound; + + /* + * For binary search: refers to the 1st and + * last uncompared elements. + */ + tmp_l_bound = lBound + 1; + tmp_r_bound = sslot.nvalues - 2; + + while (tmp_l_bound <= tmp_r_bound) + { + int mid = (tmp_l_bound + tmp_r_bound) / 2; + + compare = ApplySortComparator(constval, false, + sslot.values[mid], false, &ssup); + if (ON_RIGHT(compare)) + { + /* + * values[mid] < CONST: continue + * rightward search. + */ + rBound = mid; + tmp_l_bound = mid + 1; + + } + else if (ON_EQUAL(compare)) + { + /* values[mid] == CONST */ + if (cmptype == COMPARE_LT) + { + /* + * When cmptype is "<": values[mid + * - 1] is rightmost MCV value + * satisfying the predicate. + */ + rBound = mid - 1; + } + else if (cmptype == COMPARE_LE) + { + /* + * When cmptype is "<=": + * values[mid] is rightmost MCV + * value satisfying the predicate. + */ + rBound = mid; + } + + break; /* stop search */ + } + else if (ON_LEFT(compare)) + { + /* continue leftward search */ + tmp_r_bound = mid - 1; + } + } + } + } + } + else if (cmptype == COMPARE_GT || cmptype == COMPARE_GE) + { + /* For predicate: (VAR ">" CONST) or (VAR ">=" CONST) */ + + /* + * Start by comparing with the Max VAR: + * values[sslot.nvalues - 1] + */ + + compare = ApplySortComparator(constval, false, + sslot.values[sslot.nvalues - 1], false, &ssup); + if (ON_RIGHT(compare)) + { + /* + * CONST > Max VAR: all MCV entries < CONST, no + * MCV entries satisfy the predicate. + */ + in_mcv_range = IN_MCV_RANGE_NO; + } + else if (ON_EQUAL(compare)) + { + /* CONST == Max VAR */ + if (cmptype == COMPARE_GT) + { + /* + * When cmptype is ">": All MCV entries < + * CONST, no MCV entries satisfy the + * predicate. + */ + in_mcv_range = IN_MCV_RANGE_NO; + } + else if (cmptype == COMPARE_GE) + { + /* + * When cmptype is ">=": only the Max VAR + * satisfies the predicate. + */ + in_mcv_range = IN_MCV_RANGE_YES; + rBound = sslot.nvalues - 1; + lBound = rBound; + } + } + else if (ON_LEFT(compare)) + { + /* + * CONST < Max VAR. Right boundary determined, now + * find left boundary. + */ + in_mcv_range = IN_MCV_RANGE_YES; + rBound = sslot.nvalues - 1; + + /* compare at Min VAR: values[0] */ + compare = ApplySortComparator(constval, false, + sslot.values[0], false, &ssup); + if (ON_LEFT(compare)) + { + /* + * CONST < Min VAR: All MCV entries ">"/">=" + * CONST, all MCV entries satisfy the + * predicate. + */ + lBound = 0; + } + else if (ON_EQUAL(compare)) + { + /* CONST == Min VAR */ + if (cmptype == COMPARE_GT) + { + /* + * When cmptype is ">": values[1] is the + * leftmost entry satisfying the + * predicate. + */ + lBound = 1; + } + else if (cmptype == COMPARE_GE) + { + /* + * when cmptype is ">=": all MCV entries + * ">=" CONST, all MCV entries satisfy the + * predicate. + */ + lBound = 0; + } + } + else if (ON_RIGHT(compare)) + { + /* + * Min VAR > CONST: binary search. + */ + int tmp_l_bound; + int tmp_r_bound; + + /* + * Initially set left boundary equal to right + * boundary + */ + lBound = rBound; + + /* + * For binary search: refers to the 1st and + * last uncompared elements. + */ + tmp_l_bound = 1; + tmp_r_bound = rBound - 1; + + while (tmp_l_bound <= tmp_r_bound) + { + int mid = (tmp_l_bound + tmp_r_bound) / 2; + + compare = ApplySortComparator(constval, false, + sslot.values[mid], false, &ssup); + + if (ON_LEFT(compare)) + { + /* + * CONST < values[mid]: continue + * leftward search. + */ + lBound = mid; + tmp_r_bound = mid - 1; + } + else if (ON_EQUAL(compare)) + { + /* values[mid] == CONST */ + if (cmptype == COMPARE_GT) + { + /* + * When cmptype is ">": values[mid + * + 1] is the min one satisfy the + * predicate. + */ + lBound = mid + 1; + } + else if (cmptype == COMPARE_GE) + { + /* + * When cmptype is >=: values[mid] + * is the min one satisfy the + * predicate. + */ + lBound = mid; + } + + break; /* stop search */ + } + else if (ON_RIGHT(compare)) + { + /* continue rightward search */ + tmp_l_bound = mid + 1; + } + } + } + } + } + } + } + } for (i = 0; i < sslot.nvalues; i++) { - Datum fresult; + /* Accumulate sumcommon first */ + sumcommon += sslot.numbers[i]; - if (varonleft) - fcinfo->args[0].value = sslot.values[i]; - else - fcinfo->args[1].value = sslot.values[i]; + /* + * If outside the MCV range, no comparison needed; + * + * If already known to be within MCV bounds, only accumulate + * mcv_selec over entries in range with no extra comparisons; + * + * Otherwise, perform on-the-fly comparisons to identify matches + * and accumulate mcv_selec. + */ + + if (in_mcv_range == IN_MCV_RANGE_NO) + continue; + + if (in_mcv_range == IN_MCV_RANGE_YES) + { + if (lBound <= i && i <= rBound) + mcv_selec += sslot.numbers[i]; + + continue; + } + + fcinfo->args[arg_var].value = sslot.values[i]; fcinfo->isnull = false; fresult = FunctionCallInvoke(fcinfo); if (!fcinfo->isnull && DatumGetBool(fresult)) mcv_selec += sslot.numbers[i]; - sumcommon += sslot.numbers[i]; } + free_attstatsslot(&sslot); } @@ -1030,7 +1627,7 @@ generic_restriction_selectivity(PlannerInfo *root, Oid oproid, Oid collation, */ mcvsel = mcv_selectivity(&vardata, &opproc, collation, constval, varonleft, - &mcvsum); + &mcvsum, oproid); /* * If the histogram is large enough, see what fraction of it matches @@ -1275,9 +1872,9 @@ ineq_histogram_selectivity(PlannerInfo *root, &isdefault); /* Subtract off the number of known MCVs */ - if (get_attstatsslot(&mcvslot, vardata->statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_NUMBERS)) + if (get_attstatsslot_mcv(&mcvslot, vardata->statsTuple, + InvalidOid, + ATTSTATSSLOT_NUMBERS)) { otherdistinct -= mcvslot.nnumbers; free_attstatsslot(&mcvslot); @@ -1639,9 +2236,9 @@ booltestsel(PlannerInfo *root, BoolTestType booltesttype, Node *arg, stats = (Form_pg_statistic) GETSTRUCT(vardata.statsTuple); freq_null = stats->stanullfrac; - if (get_attstatsslot(&sslot, vardata.statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS) + if (get_attstatsslot_mcv(&sslot, vardata.statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS) && sslot.nnumbers > 0) { double freq_true; @@ -2418,6 +3015,8 @@ eqjoinsel(PG_FUNCTION_ARGS) bool get_mcv_stats; bool join_is_reversed; RelOptInfo *inner_rel; + int statskind1 = 0; + int statskind2 = 0; get_join_variables(root, args, sjinfo, &vardata1, &vardata2, &join_is_reversed); @@ -2436,12 +3035,12 @@ eqjoinsel(PG_FUNCTION_ARGS) */ get_mcv_stats = (HeapTupleIsValid(vardata1.statsTuple) && HeapTupleIsValid(vardata2.statsTuple) && - get_attstatsslot(&sslot1, vardata1.statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - 0) && - get_attstatsslot(&sslot2, vardata2.statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - 0)); + get_attstatsslot_mcv(&sslot1, vardata1.statsTuple, + InvalidOid, + 0) && + get_attstatsslot_mcv(&sslot2, vardata2.statsTuple, + InvalidOid, + 0)); if (HeapTupleIsValid(vardata1.statsTuple)) { @@ -2449,9 +3048,9 @@ eqjoinsel(PG_FUNCTION_ARGS) stats1 = (Form_pg_statistic) GETSTRUCT(vardata1.statsTuple); if (get_mcv_stats && statistic_proc_security_check(&vardata1, opfuncoid)) - have_mcvs1 = get_attstatsslot(&sslot1, vardata1.statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); + have_mcvs1 = (statskind1 = get_attstatsslot_mcv(&sslot1, vardata1.statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS)); } if (HeapTupleIsValid(vardata2.statsTuple)) @@ -2460,9 +3059,9 @@ eqjoinsel(PG_FUNCTION_ARGS) stats2 = (Form_pg_statistic) GETSTRUCT(vardata2.statsTuple); if (get_mcv_stats && statistic_proc_security_check(&vardata2, opfuncoid)) - have_mcvs2 = get_attstatsslot(&sslot2, vardata2.statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS); + have_mcvs2 = (statskind2 = get_attstatsslot_mcv(&sslot2, vardata2.statsTuple, + InvalidOid, + ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS)); } /* Prepare info usable by both eqjoinsel_inner and eqjoinsel_semi */ @@ -2494,7 +3093,8 @@ eqjoinsel(PG_FUNCTION_ARGS) stats1, stats2, have_mcvs1, have_mcvs2, hasmatch1, hasmatch2, - &nmatches); + &nmatches, + operator, statskind1, statskind2); switch (sjinfo->jointype) { @@ -2526,7 +3126,8 @@ eqjoinsel(PG_FUNCTION_ARGS) have_mcvs1, have_mcvs2, hasmatch1, hasmatch2, &nmatches, - inner_rel); + inner_rel, + operator, statskind1, statskind2); else selec = eqjoinsel_semi(&eqproc, collation, hashLeft, hashRight, @@ -2539,7 +3140,8 @@ eqjoinsel(PG_FUNCTION_ARGS) have_mcvs2, have_mcvs1, hasmatch2, hasmatch1, &nmatches, - inner_rel); + inner_rel, + operator, statskind2, statskind1); /* * We should never estimate the output of a semijoin to be more @@ -2597,7 +3199,8 @@ eqjoinsel_inner(FmgrInfo *eqproc, Oid collation, Form_pg_statistic stats1, Form_pg_statistic stats2, bool have_mcvs1, bool have_mcvs2, bool *hasmatch1, bool *hasmatch2, - int *p_nmatches) + int *p_nmatches, + Oid eqoproid, int statskind1, int statskind2) { double selec; @@ -2636,7 +3239,8 @@ eqjoinsel_inner(FmgrInfo *eqproc, Oid collation, sslot1, sslot2, sslot1->nvalues, sslot2->nvalues, hasmatch1, hasmatch2, - p_nmatches, &matchprodfreq); + p_nmatches, &matchprodfreq, + eqoproid, statskind1, statskind2); nmatches = *p_nmatches; CLAMP_PROBABILITY(matchprodfreq); @@ -2758,7 +3362,8 @@ eqjoinsel_semi(FmgrInfo *eqproc, Oid collation, bool have_mcvs1, bool have_mcvs2, bool *hasmatch1, bool *hasmatch2, int *p_nmatches, - RelOptInfo *inner_rel) + RelOptInfo *inner_rel, + Oid eqoproid, int statskind1, int statskind2) { double selec; @@ -2845,7 +3450,8 @@ eqjoinsel_semi(FmgrInfo *eqproc, Oid collation, sslot1, sslot2, sslot1->nvalues, clamped_nvalues2, hasmatch1, hasmatch2, - p_nmatches, &matchprodfreq); + p_nmatches, &matchprodfreq, + eqoproid, statskind1, statskind2); } nmatches = *p_nmatches; @@ -2922,6 +3528,8 @@ eqjoinsel_semi(FmgrInfo *eqproc, Oid collation, * sslot1, sslot2: MCV values for the lefthand and righthand inputs * nvalues1, nvalues2: number of values to be considered (can be less than * sslotN->nvalues, but not more) + * eqoproid: OID of the "=" operator + * statskind1, statskind2: statistical "kind" of sslot1 and sslot2 * Outputs: * hasmatch1[], hasmatch2[]: pre-zeroed arrays of lengths nvalues1, nvalues2; * entries are set to true if that MCV has a match on the other side @@ -2944,7 +3552,8 @@ eqjoinsel_find_matches(FmgrInfo *eqproc, Oid collation, AttStatsSlot *sslot1, AttStatsSlot *sslot2, int nvalues1, int nvalues2, bool *hasmatch1, bool *hasmatch2, - int *p_nmatches, double *p_matchprodfreq) + int *p_nmatches, double *p_matchprodfreq, + Oid eqoproid, int statskind1, int statskind2) { LOCAL_FCINFO(fcinfo, 2); double matchprodfreq = 0.0; @@ -3072,9 +3681,14 @@ eqjoinsel_find_matches(FmgrInfo *eqproc, Oid collation, } else { - /* We're not to use hashing, so do it the O(N^2) way */ + /* + * We're not to use hashing. First consider whether we can exploit the + * ordering of values[] in sslot1 and sslot2 for optimization, similar + * to what var_eq_const() does; otherwise do it the O(N^2) way. + */ int index1, index2; + bool use_sorted_mcv_match = false; /* Set up to supply the values in the order the operator expects */ if (op_is_reversed) @@ -3088,25 +3702,136 @@ eqjoinsel_find_matches(FmgrInfo *eqproc, Oid collation, index2 = 1; } - for (int i = 0; i < nvalues1; i++) + if (!op_is_reversed && + sslot1->stacoll == collation && + sslot2->stacoll == collation && + statskind2 == STATISTIC_KIND_MCV_VALUE_SORTED && + nvalues1 > MCV_SPECIAL_COMPARE_THRESHOLD && + nvalues2 > MCV_SPECIAL_COMPARE_THRESHOLD && + comparison_ops_are_compatible(sslot1->staop, sslot2->staop) && + comparison_ops_are_compatible(sslot1->staop, eqoproid) && + comparison_ops_are_compatible(sslot2->staop, eqoproid) && + sslot1->valuetype == sslot2->valuetype) { - fcinfo->args[index1].value = sslot1->values[i]; + /* + * If the operator is not reversed, the statistics and comparison + * collations match, both MCV lists have the same value type and + * are large enough, sslot2 contains sorted MCV values, and the + * MCV orderings are compatible with each other and with the + * equality operator, use the ordering of the MCV values to find + * matching MCVs efficiently. + */ + Oid ltopr; + + /* Look for default "<" operators for valuetype. */ + get_sort_group_operators(sslot1->valuetype, + false, false, false, + <opr, NULL, NULL, + NULL); + if (OidIsValid(ltopr)) + { + SortSupportData ssup = {0}; + int compare; + int lBound2 = 0; + bool compare_done = false; + + use_sorted_mcv_match = true; + + ssup.ssup_cxt = CurrentMemoryContext; + ssup.ssup_collation = sslot1->stacoll; + ssup.ssup_nulls_first = false; + ssup.abbreviate = false; + PrepareSortSupportFromOrderingOp(ltopr, &ssup); + + for (int i = 0; !compare_done && i < nvalues1; i++) + { + for (int j = lBound2; !compare_done && j < nvalues2; j++) + { + if (hasmatch2[j]) + continue; + + compare = ApplySortComparator(sslot1->values[i], false, + sslot2->values[j], false, &ssup); + if (ON_EQUAL(compare)) + { + hasmatch1[i] = hasmatch2[j] = true; + matchprodfreq += sslot1->numbers[i] * sslot2->numbers[j]; + nmatches++; + + if (statskind1 == STATISTIC_KIND_MCV_VALUE_SORTED) + { + /* + * Since sslot1's MCV values are sorted, start + * the next search from sslot2[j + 1]. + */ + lBound2 = j + 1; + } + + break; + } + else if (ON_LEFT(compare)) + { + /* + * sslot1->values[i] < sslot2->values[j], so all + * remaining values in sslot2 are greater than + * sslot1->values[i]. No match is possible for + * sslot1->values[i]. + */ + break; + } + else if (ON_RIGHT(compare)) + { + compare = ApplySortComparator(sslot1->values[i], false, + sslot2->values[nvalues2 - 1], false, &ssup); + if (ON_RIGHT(compare)) + { + /* + * sslot1->values[i] > sslot2->values[nvalues2 + * - 1], so all values in sslot2 are less than + * sslot1->values[i]. No match is possible for + * sslot1->values[i]. + */ + + if (statskind1 == STATISTIC_KIND_MCV_VALUE_SORTED) + { + /* + * Since sslot1's MCV values are sorted, + * no remaining values in sslot1 can match + * any value in sslot2. Stop comparing. + */ + compare_done = true; + } + + break; + } + } + } + } + } + } - for (int j = 0; j < nvalues2; j++) + if (!use_sorted_mcv_match) + { + for (int i = 0; i < nvalues1; i++) { - Datum fresult; + fcinfo->args[index1].value = sslot1->values[i]; - if (hasmatch2[j]) - continue; - fcinfo->args[index2].value = sslot2->values[j]; - fcinfo->isnull = false; - fresult = FunctionCallInvoke(fcinfo); - if (!fcinfo->isnull && DatumGetBool(fresult)) + for (int j = 0; j < nvalues2; j++) { - hasmatch1[i] = hasmatch2[j] = true; - matchprodfreq += sslot1->numbers[i] * sslot2->numbers[j]; - nmatches++; - break; + Datum fresult; + + if (hasmatch2[j]) + continue; + fcinfo->args[index2].value = sslot2->values[j]; + fcinfo->isnull = false; + fresult = FunctionCallInvoke(fcinfo); + if (!fcinfo->isnull && DatumGetBool(fresult)) + { + hasmatch1[i] = hasmatch2[j] = true; + matchprodfreq += sslot1->numbers[i] * sslot2->numbers[j]; + nmatches++; + break; + } } } } @@ -4429,6 +5154,7 @@ estimate_hash_bucket_stats(PlannerInfo *root, Node *hashkey, double nbuckets, ndistinct; bool isdefault; AttStatsSlot sslot; + int statskind; examine_variable(root, hashkey, 0, &vardata); @@ -4438,15 +5164,15 @@ estimate_hash_bucket_stats(PlannerInfo *root, Node *hashkey, double nbuckets, /* Look up the frequency of the most common value, if available */ if (HeapTupleIsValid(vardata.statsTuple)) { - if (get_attstatsslot(&sslot, vardata.statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - ATTSTATSSLOT_NUMBERS)) + statskind = get_attstatsslot_mcv(&sslot, vardata.statsTuple, + InvalidOid, + ATTSTATSSLOT_NUMBERS); + if (statskind) { /* - * The first MCV stat is for the most common value. + * Get the most common value. */ - if (sslot.nnumbers > 0) - *mcv_freq = sslot.numbers[0]; + (void) get_max_mcv_frequency(&sslot, statskind, mcv_freq); free_attstatsslot(&sslot); } else if (get_attstatsslot(&sslot, vardata.statsTuple, @@ -6854,6 +7580,7 @@ get_variable_range(PlannerInfo *root, VariableStatData *vardata, Oid opfuncoid; FmgrInfo opproc; AttStatsSlot sslot; + int statskind; /* * XXX It's very tempting to try to use the actual column min and max, if @@ -6918,7 +7645,8 @@ get_variable_range(PlannerInfo *root, VariableStatData *vardata, { get_stats_slot_range(&sslot, opfuncoid, &opproc, collation, typLen, typByVal, - &tmin, &tmax, &have_data); + &tmin, &tmax, &have_data, + STATISTIC_KIND_HISTOGRAM, sortop); free_attstatsslot(&sslot); } @@ -6930,10 +7658,11 @@ get_variable_range(PlannerInfo *root, VariableStatData *vardata, * data. Proceed only if the MCVs represent the whole table (to within * roundoff error). */ - if (get_attstatsslot(&sslot, vardata->statsTuple, - STATISTIC_KIND_MCV, InvalidOid, - have_data ? ATTSTATSSLOT_VALUES : - (ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS))) + statskind = get_attstatsslot_mcv(&sslot, vardata->statsTuple, + InvalidOid, + have_data ? ATTSTATSSLOT_VALUES : + (ATTSTATSSLOT_VALUES | ATTSTATSSLOT_NUMBERS)); + if (statskind) { bool use_mcvs = have_data; @@ -6953,7 +7682,9 @@ get_variable_range(PlannerInfo *root, VariableStatData *vardata, if (use_mcvs) get_stats_slot_range(&sslot, opfuncoid, &opproc, collation, typLen, typByVal, - &tmin, &tmax, &have_data); + &tmin, &tmax, &have_data, + statskind, sortop); + free_attstatsslot(&sslot); } @@ -6971,41 +7702,85 @@ get_variable_range(PlannerInfo *root, VariableStatData *vardata, static void get_stats_slot_range(AttStatsSlot *sslot, Oid opfuncoid, FmgrInfo *opproc, Oid collation, int16 typLen, bool typByVal, - Datum *min, Datum *max, bool *p_have_data) + Datum *min, Datum *max, bool *p_have_data, + int statskind, Oid sortop) { Datum tmin = *min; Datum tmax = *max; bool have_data = *p_have_data; bool found_tmin = false; bool found_tmax = false; + bool used_sorted_mcv = false; /* Look up the comparison function, if we didn't already do so */ if (opproc->fn_oid != opfuncoid) fmgr_info(opfuncoid, opproc); - /* Scan all the slot's values */ - for (int i = 0; i < sslot->nvalues; i++) + if (sslot->stacoll == collation && + statskind == STATISTIC_KIND_MCV_VALUE_SORTED && + sslot->nvalues > 0 && + comparison_ops_are_compatible(sslot->staop, sortop)) { + /* + * Like the STATISTIC_KIND_HISTOGRAM case in get_variable_range(), use + * the first and last values of a STATISTIC_KIND_MCV_VALUE_SORTED slot + * directly when its ordering is compatible with sortop and the + * collations match. + */ + used_sorted_mcv = true; + if (!have_data) { - tmin = tmax = sslot->values[i]; + tmin = sslot->values[0]; + tmax = sslot->values[sslot->nvalues - 1]; found_tmin = found_tmax = true; *p_have_data = have_data = true; - continue; } - if (DatumGetBool(FunctionCall2Coll(opproc, - collation, - sslot->values[i], tmin))) + else { - tmin = sslot->values[i]; - found_tmin = true; + if (DatumGetBool(FunctionCall2Coll(opproc, + collation, + sslot->values[0], tmin))) + { + tmin = sslot->values[0]; + found_tmin = true; + } + if (DatumGetBool(FunctionCall2Coll(opproc, + collation, + tmax, sslot->values[sslot->nvalues - 1]))) + { + tmax = sslot->values[sslot->nvalues - 1]; + found_tmax = true; + } } - if (DatumGetBool(FunctionCall2Coll(opproc, - collation, - tmax, sslot->values[i]))) + } + + if (!used_sorted_mcv) + { + /* Scan all the slot's values */ + for (int i = 0; i < sslot->nvalues; i++) { - tmax = sslot->values[i]; - found_tmax = true; + if (!have_data) + { + tmin = tmax = sslot->values[i]; + found_tmin = found_tmax = true; + *p_have_data = have_data = true; + continue; + } + if (DatumGetBool(FunctionCall2Coll(opproc, + collation, + sslot->values[i], tmin))) + { + tmin = sslot->values[i]; + found_tmin = true; + } + if (DatumGetBool(FunctionCall2Coll(opproc, + collation, + tmax, sslot->values[i]))) + { + tmax = sslot->values[i]; + found_tmax = true; + } } } diff --git a/src/backend/utils/cache/lsyscache.c b/src/backend/utils/cache/lsyscache.c index 65b2d64..8eaf4a1 100644 --- a/src/backend/utils/cache/lsyscache.c +++ b/src/backend/utils/cache/lsyscache.c @@ -3650,6 +3650,28 @@ get_attstatsslot(AttStatsSlot *sslot, HeapTuple statstuple, return true; } +/* + * Try to fetch the value-sorted MCV statistics first, and fall back + * to the traditional MCV statistics for compatibility with statistics + * generated without STATISTIC_KIND_MCV_VALUE_SORTED. + */ +int +get_attstatsslot_mcv(AttStatsSlot *sslot, HeapTuple statstuple, + Oid reqop, int flags) +{ + if (get_attstatsslot(sslot, statstuple, + STATISTIC_KIND_MCV_VALUE_SORTED, reqop, + flags)) + return STATISTIC_KIND_MCV_VALUE_SORTED; + + if (get_attstatsslot(sslot, statstuple, + STATISTIC_KIND_MCV, reqop, + flags)) + return STATISTIC_KIND_MCV; + + return 0; +} + /* * free_attstatsslot * Free data allocated by get_attstatsslot diff --git a/src/include/catalog/pg_statistic.h b/src/include/catalog/pg_statistic.h index 6404b2b..6d09ec4 100644 --- a/src/include/catalog/pg_statistic.h +++ b/src/include/catalog/pg_statistic.h @@ -291,6 +291,12 @@ DECLARE_FOREIGN_KEY((starelid, staattnum), pg_attribute, (attrelid, attnum)); */ #define STATISTIC_KIND_BOUNDS_HISTOGRAM 7 +/* + * Like STATISTIC_KIND_MCV, except that MCV values and their + * corresponding frequencies are sorted in ascending value order. + */ +#define STATISTIC_KIND_MCV_VALUE_SORTED 8 + #endif /* EXPOSE_TO_CLIENT_CODE */ #endif /* PG_STATISTIC_H */ diff --git a/src/include/statistics/stat_utils.h b/src/include/statistics/stat_utils.h index 957bddc..b6e0f68 100644 --- a/src/include/statistics/stat_utils.h +++ b/src/include/statistics/stat_utils.h @@ -15,6 +15,7 @@ #include "access/attnum.h" #include "fmgr.h" +#include "utils/lsyscache.h" /* avoid including primnodes.h here */ typedef struct RangeVar RangeVar; @@ -65,4 +66,8 @@ extern bool statatt_get_range_type(TypeCacheEntry *basetypcache, extern bool statatt_check_bounds_histogram(Datum arrayval); +extern bool get_max_mcv_frequency(AttStatsSlot *sslot, int statskind, + double *max_frequency); +extern bool get_min_mcv_frequency(AttStatsSlot *sslot, int statskind, + double *min_frequency); #endif /* STATS_UTILS_H */ diff --git a/src/include/statistics/statistics.h b/src/include/statistics/statistics.h index 0b16310..d6f64ca 100644 --- a/src/include/statistics/statistics.h +++ b/src/include/statistics/statistics.h @@ -149,7 +149,8 @@ extern bool import_attribute_statistics(Relation rel, const NullableDatum *elem_count_histogram, const NullableDatum *range_length_histogram, const NullableDatum *range_empty_frac, - const NullableDatum *range_bounds_histogram); + const NullableDatum *range_bounds_histogram, + const NullableDatum *most_common_vals_kind); extern bool delete_attribute_statistics(Relation rel, AttrNumber attnum, bool inherited); diff --git a/src/include/utils/lsyscache.h b/src/include/utils/lsyscache.h index 171fbda..213f283 100644 --- a/src/include/utils/lsyscache.h +++ b/src/include/utils/lsyscache.h @@ -199,6 +199,8 @@ extern int32 get_typavgwidth(Oid typid, int32 typmod); extern int32 get_attavgwidth(Oid relid, AttrNumber attnum); extern bool get_attstatsslot(AttStatsSlot *sslot, HeapTuple statstuple, int reqkind, Oid reqop, int flags); +extern int get_attstatsslot_mcv(AttStatsSlot *sslot, HeapTuple statstuple, + Oid reqop, int flags); extern void free_attstatsslot(AttStatsSlot *sslot); extern char *get_namespace_name(Oid nspid); extern char *get_namespace_name_or_temp(Oid nspid); diff --git a/src/include/utils/selfuncs.h b/src/include/utils/selfuncs.h index 8d9fff9..e488735 100644 --- a/src/include/utils/selfuncs.h +++ b/src/include/utils/selfuncs.h @@ -177,7 +177,7 @@ extern double get_variable_numdistinct(VariableStatData *vardata, extern double mcv_selectivity(VariableStatData *vardata, FmgrInfo *opproc, Oid collation, Datum constval, bool varonleft, - double *sumcommonp); + double *sumcommonp, Oid operator); extern double histogram_selectivity(VariableStatData *vardata, FmgrInfo *opproc, Oid collation, Datum constval, bool varonleft, diff --git a/src/test/regress/expected/rules.out b/src/test/regress/expected/rules.out index 0addd04..90e2b49 100644 --- a/src/test/regress/expected/rules.out +++ b/src/test/regress/expected/rules.out @@ -2633,6 +2633,11 @@ pg_stats| SELECT n.nspname AS schemaname, WHEN (s.stakind3 = 1) THEN s.stavalues3 WHEN (s.stakind4 = 1) THEN s.stavalues4 WHEN (s.stakind5 = 1) THEN s.stavalues5 + WHEN (s.stakind1 = 8) THEN s.stavalues1 + WHEN (s.stakind2 = 8) THEN s.stavalues2 + WHEN (s.stakind3 = 8) THEN s.stavalues3 + WHEN (s.stakind4 = 8) THEN s.stavalues4 + WHEN (s.stakind5 = 8) THEN s.stavalues5 ELSE NULL::anyarray END AS most_common_vals, CASE @@ -2641,6 +2646,11 @@ pg_stats| SELECT n.nspname AS schemaname, WHEN (s.stakind3 = 1) THEN s.stanumbers3 WHEN (s.stakind4 = 1) THEN s.stanumbers4 WHEN (s.stakind5 = 1) THEN s.stanumbers5 + WHEN (s.stakind1 = 8) THEN s.stanumbers1 + WHEN (s.stakind2 = 8) THEN s.stanumbers2 + WHEN (s.stakind3 = 8) THEN s.stanumbers3 + WHEN (s.stakind4 = 8) THEN s.stanumbers4 + WHEN (s.stakind5 = 8) THEN s.stanumbers5 ELSE NULL::real[] END AS most_common_freqs, CASE @@ -2760,6 +2770,11 @@ pg_stats_ext_exprs| SELECT cn.nspname AS schemaname, WHEN ((stat.a).stakind3 = 1) THEN (stat.a).stavalues3 WHEN ((stat.a).stakind4 = 1) THEN (stat.a).stavalues4 WHEN ((stat.a).stakind5 = 1) THEN (stat.a).stavalues5 + WHEN ((stat.a).stakind1 = 8) THEN (stat.a).stavalues1 + WHEN ((stat.a).stakind2 = 8) THEN (stat.a).stavalues2 + WHEN ((stat.a).stakind3 = 8) THEN (stat.a).stavalues3 + WHEN ((stat.a).stakind4 = 8) THEN (stat.a).stavalues4 + WHEN ((stat.a).stakind5 = 8) THEN (stat.a).stavalues5 ELSE NULL::anyarray END AS most_common_vals, CASE @@ -2768,6 +2783,11 @@ pg_stats_ext_exprs| SELECT cn.nspname AS schemaname, WHEN ((stat.a).stakind3 = 1) THEN (stat.a).stanumbers3 WHEN ((stat.a).stakind4 = 1) THEN (stat.a).stanumbers4 WHEN ((stat.a).stakind5 = 1) THEN (stat.a).stanumbers5 + WHEN ((stat.a).stakind1 = 8) THEN (stat.a).stanumbers1 + WHEN ((stat.a).stakind2 = 8) THEN (stat.a).stanumbers2 + WHEN ((stat.a).stakind3 = 8) THEN (stat.a).stanumbers3 + WHEN ((stat.a).stakind4 = 8) THEN (stat.a).stanumbers4 + WHEN ((stat.a).stakind5 = 8) THEN (stat.a).stanumbers5 ELSE NULL::real[] END AS most_common_freqs, CASE diff --git a/src/test/regress/expected/stats_import.out b/src/test/regress/expected/stats_import.out index 87b0e03..c814593 100644 --- a/src/test/regress/expected/stats_import.out +++ b/src/test/regress/expected/stats_import.out @@ -1769,6 +1769,7 @@ CREATE STATISTICS stats_import.test_stat_clone -- -- Copy stats from test to test_clone, and is_odd to is_odd_clone -- +-- If most_common_vals_kind is not specified, default to 1-STATISTIC_KIND_MCV. SELECT s.schemaname, s.tablename, s.attname, s.inherited, r.* FROM pg_catalog.pg_stats AS s CROSS JOIN LATERAL @@ -1824,6 +1825,131 @@ FROM stats_import.pg_statistic_get_difference('test', 'test_clone') \gx (0 rows) +SELECT relname, (stats).* +FROM stats_import.pg_statistic_get_difference('is_odd', 'is_odd_clone') +\gx +-[ RECORD 1 ]------------- +relname | is_odd +attname | comp_2_1 +stainherit | f +stanullfrac | 0.25 +stawidth | 1 +stadistinct | -0.5 +stakind1 | 8 +stakind2 | 3 +stakind3 | 0 +stakind4 | 0 +stakind5 | 0 +staop1 | 91 +staop2 | 58 +staop3 | 0 +staop4 | 0 +staop5 | 0 +stacoll1 | 0 +stacoll2 | 0 +stacoll3 | 0 +stacoll4 | 0 +stacoll5 | 0 +stanumbers1 | {0.5} +stanumbers2 | {0.5} +stanumbers3 | +stanumbers4 | +stanumbers5 | +sv1 | {t} +sv2 | +sv3 | +sv4 | +sv5 | +-[ RECORD 2 ]------------- +relname | is_odd_clone +attname | comp_2_1 +stainherit | f +stanullfrac | 0.25 +stawidth | 1 +stadistinct | -0.5 +stakind1 | 1 +stakind2 | 3 +stakind3 | 0 +stakind4 | 0 +stakind5 | 0 +staop1 | 91 +staop2 | 58 +staop3 | 0 +staop4 | 0 +staop5 | 0 +stacoll1 | 0 +stacoll2 | 0 +stacoll3 | 0 +stacoll4 | 0 +stacoll5 | 0 +stanumbers1 | {0.5} +stanumbers2 | {0.5} +stanumbers3 | +stanumbers4 | +stanumbers5 | +sv1 | {t} +sv2 | +sv3 | +sv4 | +sv5 | + +-- Specify 8-STATISTIC_KIND_MCV_VALUE_SORTED. +SELECT s.schemaname, s.tablename, s.attname, s.inherited, r.* +FROM pg_catalog.pg_stats AS s +CROSS JOIN LATERAL + pg_catalog.pg_restore_attribute_stats( + 'schemaname', 'stats_import', + 'relname', s.tablename::text || '_clone', + 'attname', s.attname::text, + 'inherited', s.inherited, + 'version', 150000, + 'null_frac', s.null_frac, + 'avg_width', s.avg_width, + 'n_distinct', s.n_distinct, + 'most_common_vals', s.most_common_vals::text, + 'most_common_freqs', s.most_common_freqs, + 'histogram_bounds', s.histogram_bounds::text, + 'correlation', s.correlation, + 'most_common_elems', s.most_common_elems::text, + 'most_common_elem_freqs', s.most_common_elem_freqs, + 'elem_count_histogram', s.elem_count_histogram, + 'range_bounds_histogram', s.range_bounds_histogram::text, + 'range_empty_frac', s.range_empty_frac, + 'range_length_histogram', s.range_length_histogram::text, + 'most_common_vals_kind', 8::smallint) AS r +WHERE s.schemaname = 'stats_import' +AND s.tablename IN ('test', 'is_odd') +ORDER BY s.tablename, s.attname, s.inherited; + schemaname | tablename | attname | inherited | r +--------------+-----------+----------+-----------+--- + stats_import | is_odd | comp_2_1 | f | t + stats_import | test | arange | f | t + stats_import | test | comp | f | t + stats_import | test | id | f | t + stats_import | test | name | f | t + stats_import | test | tags | f | t +(6 rows) + +SELECT c.relname, COUNT(*) AS num_stats +FROM pg_class AS c +JOIN pg_statistic s ON s.starelid = c.oid +WHERE c.relnamespace = 'stats_import'::regnamespace +AND c.relname IN ('test', 'test_clone', 'is_odd', 'is_odd_clone') +GROUP BY c.relname +ORDER BY c.relname; + relname | num_stats +--------------+----------- + is_odd | 1 + is_odd_clone | 1 + test | 5 + test_clone | 5 +(4 rows) + +SELECT relname, (stats).* +FROM stats_import.pg_statistic_get_difference('test', 'test_clone') +\gx +(0 rows) + SELECT relname, (stats).* FROM stats_import.pg_statistic_get_difference('is_odd', 'is_odd_clone') \gx @@ -4149,6 +4275,550 @@ SELECT COUNT(*) FROM stats_import.test_range_expr_null 19 (1 row) +-- Test pg_restore_extended_stats() with respect to the most_common_vals_kind parameter. +-- If most_common_vals_kind is not specified, default to 1-STATISTIC_KIND_MCV. +SELECT pg_catalog.pg_restore_extended_stats( + 'schemaname', 'stats_import', + 'relname', 'test_clone', + 'statistics_schemaname', 'stats_import', + 'statistics_name', 'test_stat_clone', + 'inherited', false, + 'exprs', '[ + { "most_common_vals": "{1,2,3}", "most_common_freqs": "{0.1,0.2,0.3}" }, + { "most_common_vals": "{11,22,33}", "most_common_freqs": "{0.11,0.22,0.33}" } + ]'::jsonb); + pg_restore_extended_stats +--------------------------- + t +(1 row) + +SELECT cn.nspname AS schemaname, + c.relname AS tablename, + s.stxrelid AS tableid, + sn.nspname AS statistics_schemaname, + s.stxname AS statistics_name, + s.oid AS statistics_id, + pg_get_userbyid(s.stxowner) AS statistics_owner, + stat.expr, + sd.stxdinherit AS inherited, + (stat.a).stanullfrac AS null_frac, + (stat.a).stawidth AS avg_width, + (stat.a).stadistinct AS n_distinct, + (CASE + WHEN (stat.a).stakind1 = 1 THEN 1 + WHEN (stat.a).stakind2 = 1 THEN 1 + WHEN (stat.a).stakind3 = 1 THEN 1 + WHEN (stat.a).stakind4 = 1 THEN 1 + WHEN (stat.a).stakind5 = 1 THEN 1 + WHEN (stat.a).stakind1 = 8 THEN 8 + WHEN (stat.a).stakind2 = 8 THEN 8 + WHEN (stat.a).stakind3 = 8 THEN 8 + WHEN (stat.a).stakind4 = 8 THEN 8 + WHEN (stat.a).stakind5 = 8 THEN 8 + else -1 + END) AS most_common_vals_kind, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stavalues5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stavalues5 + END) AS most_common_vals, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stanumbers5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stanumbers5 + END) AS most_common_freqs, + (CASE + WHEN (stat.a).stakind1 = 2 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 2 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 2 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 2 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 2 THEN (stat.a).stavalues5 + END) AS histogram_bounds, + (CASE + WHEN (stat.a).stakind1 = 3 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 3 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 3 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 3 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 3 THEN (stat.a).stanumbers5[1] + END) correlation, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx +-[ RECORD 1 ]----------+---------------------- +schemaname | stats_import +tablename | test_clone +tableid | 17667 +statistics_schemaname | stats_import +statistics_name | test_stat_clone +statistics_id | 17674 +statistics_owner | xman +expr | lower(arange) +inherited | f +null_frac | 0 +avg_width | 0 +n_distinct | 0 +most_common_vals_kind | 1 +most_common_vals | {1,2,3} +most_common_freqs | {0.1,0.2,0.3} +histogram_bounds | +correlation | +most_common_elems | +most_common_elem_freqs | +elem_count_histogram | +range_length_histogram | +range_empty_frac | +range_bounds_histogram | +-[ RECORD 2 ]----------+---------------------- +schemaname | stats_import +tablename | test_clone +tableid | 17667 +statistics_schemaname | stats_import +statistics_name | test_stat_clone +statistics_id | 17674 +statistics_owner | xman +expr | array_length(tags, 1) +inherited | f +null_frac | 0 +avg_width | 0 +n_distinct | 0 +most_common_vals_kind | 1 +most_common_vals | {11,22,33} +most_common_freqs | {0.11,0.22,0.33} +histogram_bounds | +correlation | +most_common_elems | +most_common_elem_freqs | +elem_count_histogram | +range_length_histogram | +range_empty_frac | +range_bounds_histogram | + +-- Specify most_common_vals_kind as 1-STATISTIC_KIND_MCV. +SELECT pg_catalog.pg_restore_extended_stats( + 'schemaname', 'stats_import', + 'relname', 'test_clone', + 'statistics_schemaname', 'stats_import', + 'statistics_name', 'test_stat_clone', + 'inherited', false, + 'exprs', '[ + { "most_common_vals": "{11,22,33}", "most_common_freqs": "{0.11,0.22,0.3}", "most_common_vals_kind":"1" }, + { "most_common_vals": "{101,202,302}", "most_common_freqs": "{0.101,0.202,0.303}" } + ]'::jsonb); + pg_restore_extended_stats +--------------------------- + t +(1 row) + +SELECT cn.nspname AS schemaname, + c.relname AS tablename, + s.stxrelid AS tableid, + sn.nspname AS statistics_schemaname, + s.stxname AS statistics_name, + s.oid AS statistics_id, + pg_get_userbyid(s.stxowner) AS statistics_owner, + stat.expr, + sd.stxdinherit AS inherited, + (stat.a).stanullfrac AS null_frac, + (stat.a).stawidth AS avg_width, + (stat.a).stadistinct AS n_distinct, + (CASE + WHEN (stat.a).stakind1 = 1 THEN 1 + WHEN (stat.a).stakind2 = 1 THEN 1 + WHEN (stat.a).stakind3 = 1 THEN 1 + WHEN (stat.a).stakind4 = 1 THEN 1 + WHEN (stat.a).stakind5 = 1 THEN 1 + WHEN (stat.a).stakind1 = 8 THEN 8 + WHEN (stat.a).stakind2 = 8 THEN 8 + WHEN (stat.a).stakind3 = 8 THEN 8 + WHEN (stat.a).stakind4 = 8 THEN 8 + WHEN (stat.a).stakind5 = 8 THEN 8 + else -1 + END) AS most_common_vals_kind, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stavalues5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stavalues5 + END) AS most_common_vals, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stanumbers5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stanumbers5 + END) AS most_common_freqs, + (CASE + WHEN (stat.a).stakind1 = 2 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 2 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 2 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 2 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 2 THEN (stat.a).stavalues5 + END) AS histogram_bounds, + (CASE + WHEN (stat.a).stakind1 = 3 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 3 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 3 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 3 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 3 THEN (stat.a).stanumbers5[1] + END) correlation, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx +-[ RECORD 1 ]----------+---------------------- +schemaname | stats_import +tablename | test_clone +tableid | 17667 +statistics_schemaname | stats_import +statistics_name | test_stat_clone +statistics_id | 17674 +statistics_owner | xman +expr | lower(arange) +inherited | f +null_frac | 0 +avg_width | 0 +n_distinct | 0 +most_common_vals_kind | 1 +most_common_vals | {11,22,33} +most_common_freqs | {0.11,0.22,0.3} +histogram_bounds | +correlation | +most_common_elems | +most_common_elem_freqs | +elem_count_histogram | +range_length_histogram | +range_empty_frac | +range_bounds_histogram | +-[ RECORD 2 ]----------+---------------------- +schemaname | stats_import +tablename | test_clone +tableid | 17667 +statistics_schemaname | stats_import +statistics_name | test_stat_clone +statistics_id | 17674 +statistics_owner | xman +expr | array_length(tags, 1) +inherited | f +null_frac | 0 +avg_width | 0 +n_distinct | 0 +most_common_vals_kind | 1 +most_common_vals | {101,202,302} +most_common_freqs | {0.101,0.202,0.303} +histogram_bounds | +correlation | +most_common_elems | +most_common_elem_freqs | +elem_count_histogram | +range_length_histogram | +range_empty_frac | +range_bounds_histogram | + +-- Specify most_common_vals_kind as 8-STATISTIC_KIND_MCV_VALUE_SORTED. +SELECT pg_catalog.pg_restore_extended_stats( + 'schemaname', 'stats_import', + 'relname', 'test_clone', + 'statistics_schemaname', 'stats_import', + 'statistics_name', 'test_stat_clone', + 'inherited', false, + 'exprs', '[ + { "most_common_vals": "{11,12,13}", "most_common_freqs": "{0.11,0.12,0.13}", "most_common_vals_kind":"8" }, + { "most_common_vals": "{41,42,43}", "most_common_freqs": "{0.011,0.022,0.033}","most_common_vals_kind":"8" } + ]'::jsonb); + pg_restore_extended_stats +--------------------------- + t +(1 row) + +SELECT cn.nspname AS schemaname, + c.relname AS tablename, + s.stxrelid AS tableid, + sn.nspname AS statistics_schemaname, + s.stxname AS statistics_name, + s.oid AS statistics_id, + pg_get_userbyid(s.stxowner) AS statistics_owner, + stat.expr, + sd.stxdinherit AS inherited, + (stat.a).stanullfrac AS null_frac, + (stat.a).stawidth AS avg_width, + (stat.a).stadistinct AS n_distinct, + (CASE + WHEN (stat.a).stakind1 = 1 THEN 1 + WHEN (stat.a).stakind2 = 1 THEN 1 + WHEN (stat.a).stakind3 = 1 THEN 1 + WHEN (stat.a).stakind4 = 1 THEN 1 + WHEN (stat.a).stakind5 = 1 THEN 1 + WHEN (stat.a).stakind1 = 8 THEN 8 + WHEN (stat.a).stakind2 = 8 THEN 8 + WHEN (stat.a).stakind3 = 8 THEN 8 + WHEN (stat.a).stakind4 = 8 THEN 8 + WHEN (stat.a).stakind5 = 8 THEN 8 + else -1 + END) AS most_common_vals_kind, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stavalues5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stavalues5 + END) AS most_common_vals, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stanumbers5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stanumbers5 + END) AS most_common_freqs, + (CASE + WHEN (stat.a).stakind1 = 2 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 2 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 2 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 2 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 2 THEN (stat.a).stavalues5 + END) AS histogram_bounds, + (CASE + WHEN (stat.a).stakind1 = 3 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 3 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 3 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 3 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 3 THEN (stat.a).stanumbers5[1] + END) correlation, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx +-[ RECORD 1 ]----------+---------------------- +schemaname | stats_import +tablename | test_clone +tableid | 17667 +statistics_schemaname | stats_import +statistics_name | test_stat_clone +statistics_id | 17674 +statistics_owner | xman +expr | lower(arange) +inherited | f +null_frac | 0 +avg_width | 0 +n_distinct | 0 +most_common_vals_kind | 8 +most_common_vals | {11,12,13} +most_common_freqs | {0.11,0.12,0.13} +histogram_bounds | +correlation | +most_common_elems | +most_common_elem_freqs | +elem_count_histogram | +range_length_histogram | +range_empty_frac | +range_bounds_histogram | +-[ RECORD 2 ]----------+---------------------- +schemaname | stats_import +tablename | test_clone +tableid | 17667 +statistics_schemaname | stats_import +statistics_name | test_stat_clone +statistics_id | 17674 +statistics_owner | xman +expr | array_length(tags, 1) +inherited | f +null_frac | 0 +avg_width | 0 +n_distinct | 0 +most_common_vals_kind | 8 +most_common_vals | {41,42,43} +most_common_freqs | {0.011,0.022,0.033} +histogram_bounds | +correlation | +most_common_elems | +most_common_elem_freqs | +elem_count_histogram | +range_length_histogram | +range_empty_frac | +range_bounds_histogram | + DROP SCHEMA stats_import CASCADE; NOTICE: drop cascades to 27 other objects DETAIL: drop cascades to view stats_import.pg_stats_stable @@ -4178,3 +4848,189 @@ drop cascades to table stats_import.test_dom_ts_clone drop cascades to table stats_import.test_clone drop cascades to table stats_import.test_mr_clone drop cascades to table stats_import.test_range_expr_null + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx +ERROR: syntax error at or near "WHEN" +LINE 1: WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + ^ +-- Specify most_common_vals_kind as 8-STATISTIC_KIND_MCV_VALUE_SORTED. +SELECT pg_catalog.pg_restore_extended_stats( + 'schemaname', 'stats_import', + 'relname', 'test_clone', + 'statistics_schemaname', 'stats_import', + 'statistics_name', 'test_stat_clone', + 'inherited', false, + 'exprs', '[ + { "most_common_vals": "{11,12,13}", "most_common_freqs": "{0.11,0.12,0.13}", "most_common_vals_kind":"8" }, + { "most_common_vals": "{41,42,43}", "most_common_freqs": "{0.011,0.022,0.033}","most_common_vals_kind":"8" } + ]'::jsonb); +ERROR: schema "stats_import" does not exist +SELECT cn.nspname AS schemaname, + c.relname AS tablename, + s.stxrelid AS tableid, + sn.nspname AS statistics_schemaname, + s.stxname AS statistics_name, + s.oid AS statistics_id, + pg_get_userbyid(s.stxowner) AS statistics_owner, + stat.expr, + sd.stxdinherit AS inherited, + (stat.a).stanullfrac AS null_frac, + (stat.a).stawidth AS avg_width, + (stat.a).stadistinct AS n_distinct, + (CASE + WHEN (stat.a).stakind1 = 1 THEN 1 + WHEN (stat.a).stakind2 = 1 THEN 1 + WHEN (stat.a).stakind3 = 1 THEN 1 + WHEN (stat.a).stakind4 = 1 THEN 1 + WHEN (stat.a).stakind5 = 1 THEN 1 + WHEN (stat.a).stakind1 = 8 THEN 8 + WHEN (stat.a).stakind2 = 8 THEN 8 + WHEN (stat.a).stakind3 = 8 THEN 8 + WHEN (stat.a).stakind4 = 8 THEN 8 + WHEN (stat.a).stakind5 = 8 THEN 8 + else -1 + END) AS most_common_vals_kind, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stavalues5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stavalues5 + END) AS most_common_vals, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stanumbers5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stanumbers5 + END) AS most_common_freqs, + (CASE + WHEN (stat.a).stakind1 = 2 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 2 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 2 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 2 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 2 THEN (stat.a).stavalues5 + END) AS histogram_bounds, + (CASE + WHEN (stat.a).stakind1 = 3 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 3 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 3 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 3 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 3 THEN (stat.a).stanumbers5[1] + END) correlation, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx +(0 rows) + +DROP SCHEMA stats_import CASCADE; +ERROR: schema "stats_import" does not exist diff --git a/src/test/regress/sql/stats_import.sql b/src/test/regress/sql/stats_import.sql index f24da5b..484e870 100644 --- a/src/test/regress/sql/stats_import.sql +++ b/src/test/regress/sql/stats_import.sql @@ -1326,6 +1326,7 @@ CREATE STATISTICS stats_import.test_stat_clone -- -- Copy stats from test to test_clone, and is_odd to is_odd_clone -- +-- If most_common_vals_kind is not specified, default to 1-STATISTIC_KIND_MCV. SELECT s.schemaname, s.tablename, s.attname, s.inherited, r.* FROM pg_catalog.pg_stats AS s CROSS JOIN LATERAL @@ -1368,6 +1369,50 @@ SELECT relname, (stats).* FROM stats_import.pg_statistic_get_difference('is_odd', 'is_odd_clone') \gx +-- Specify 8-STATISTIC_KIND_MCV_VALUE_SORTED. +SELECT s.schemaname, s.tablename, s.attname, s.inherited, r.* +FROM pg_catalog.pg_stats AS s +CROSS JOIN LATERAL + pg_catalog.pg_restore_attribute_stats( + 'schemaname', 'stats_import', + 'relname', s.tablename::text || '_clone', + 'attname', s.attname::text, + 'inherited', s.inherited, + 'version', 150000, + 'null_frac', s.null_frac, + 'avg_width', s.avg_width, + 'n_distinct', s.n_distinct, + 'most_common_vals', s.most_common_vals::text, + 'most_common_freqs', s.most_common_freqs, + 'histogram_bounds', s.histogram_bounds::text, + 'correlation', s.correlation, + 'most_common_elems', s.most_common_elems::text, + 'most_common_elem_freqs', s.most_common_elem_freqs, + 'elem_count_histogram', s.elem_count_histogram, + 'range_bounds_histogram', s.range_bounds_histogram::text, + 'range_empty_frac', s.range_empty_frac, + 'range_length_histogram', s.range_length_histogram::text, + 'most_common_vals_kind', 8::smallint) AS r +WHERE s.schemaname = 'stats_import' +AND s.tablename IN ('test', 'is_odd') +ORDER BY s.tablename, s.attname, s.inherited; + +SELECT c.relname, COUNT(*) AS num_stats +FROM pg_class AS c +JOIN pg_statistic s ON s.starelid = c.oid +WHERE c.relnamespace = 'stats_import'::regnamespace +AND c.relname IN ('test', 'test_clone', 'is_odd', 'is_odd_clone') +GROUP BY c.relname +ORDER BY c.relname; + +SELECT relname, (stats).* +FROM stats_import.pg_statistic_get_difference('test', 'test_clone') +\gx + +SELECT relname, (stats).* +FROM stats_import.pg_statistic_get_difference('is_odd', 'is_odd_clone') +\gx + -- attribute stats exist before a clear, but not after SELECT COUNT(*) FROM pg_stats @@ -2898,4 +2943,574 @@ SELECT * FROM stats_import.test_range_expr_null SELECT COUNT(*) FROM stats_import.test_range_expr_null WHERE (rng * int4range(50, 150)) && '[60,70)'::int4range; +-- Test pg_restore_extended_stats() with respect to the most_common_vals_kind parameter. +-- If most_common_vals_kind is not specified, default to 1-STATISTIC_KIND_MCV. +SELECT pg_catalog.pg_restore_extended_stats( + 'schemaname', 'stats_import', + 'relname', 'test_clone', + 'statistics_schemaname', 'stats_import', + 'statistics_name', 'test_stat_clone', + 'inherited', false, + 'exprs', '[ + { "most_common_vals": "{1,2,3}", "most_common_freqs": "{0.1,0.2,0.3}" }, + { "most_common_vals": "{11,22,33}", "most_common_freqs": "{0.11,0.22,0.33}" } + ]'::jsonb); + +SELECT cn.nspname AS schemaname, + c.relname AS tablename, + s.stxrelid AS tableid, + sn.nspname AS statistics_schemaname, + s.stxname AS statistics_name, + s.oid AS statistics_id, + pg_get_userbyid(s.stxowner) AS statistics_owner, + stat.expr, + sd.stxdinherit AS inherited, + (stat.a).stanullfrac AS null_frac, + (stat.a).stawidth AS avg_width, + (stat.a).stadistinct AS n_distinct, + (CASE + WHEN (stat.a).stakind1 = 1 THEN 1 + WHEN (stat.a).stakind2 = 1 THEN 1 + WHEN (stat.a).stakind3 = 1 THEN 1 + WHEN (stat.a).stakind4 = 1 THEN 1 + WHEN (stat.a).stakind5 = 1 THEN 1 + WHEN (stat.a).stakind1 = 8 THEN 8 + WHEN (stat.a).stakind2 = 8 THEN 8 + WHEN (stat.a).stakind3 = 8 THEN 8 + WHEN (stat.a).stakind4 = 8 THEN 8 + WHEN (stat.a).stakind5 = 8 THEN 8 + else -1 + END) AS most_common_vals_kind, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stavalues5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stavalues5 + END) AS most_common_vals, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stanumbers5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stanumbers5 + END) AS most_common_freqs, + (CASE + WHEN (stat.a).stakind1 = 2 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 2 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 2 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 2 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 2 THEN (stat.a).stavalues5 + END) AS histogram_bounds, + (CASE + WHEN (stat.a).stakind1 = 3 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 3 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 3 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 3 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 3 THEN (stat.a).stanumbers5[1] + END) correlation, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx + +-- Specify most_common_vals_kind as 1-STATISTIC_KIND_MCV. +SELECT pg_catalog.pg_restore_extended_stats( + 'schemaname', 'stats_import', + 'relname', 'test_clone', + 'statistics_schemaname', 'stats_import', + 'statistics_name', 'test_stat_clone', + 'inherited', false, + 'exprs', '[ + { "most_common_vals": "{11,22,33}", "most_common_freqs": "{0.11,0.22,0.3}", "most_common_vals_kind":"1" }, + { "most_common_vals": "{101,202,302}", "most_common_freqs": "{0.101,0.202,0.303}" } + ]'::jsonb); + +SELECT cn.nspname AS schemaname, + c.relname AS tablename, + s.stxrelid AS tableid, + sn.nspname AS statistics_schemaname, + s.stxname AS statistics_name, + s.oid AS statistics_id, + pg_get_userbyid(s.stxowner) AS statistics_owner, + stat.expr, + sd.stxdinherit AS inherited, + (stat.a).stanullfrac AS null_frac, + (stat.a).stawidth AS avg_width, + (stat.a).stadistinct AS n_distinct, + (CASE + WHEN (stat.a).stakind1 = 1 THEN 1 + WHEN (stat.a).stakind2 = 1 THEN 1 + WHEN (stat.a).stakind3 = 1 THEN 1 + WHEN (stat.a).stakind4 = 1 THEN 1 + WHEN (stat.a).stakind5 = 1 THEN 1 + WHEN (stat.a).stakind1 = 8 THEN 8 + WHEN (stat.a).stakind2 = 8 THEN 8 + WHEN (stat.a).stakind3 = 8 THEN 8 + WHEN (stat.a).stakind4 = 8 THEN 8 + WHEN (stat.a).stakind5 = 8 THEN 8 + else -1 + END) AS most_common_vals_kind, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stavalues5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stavalues5 + END) AS most_common_vals, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stanumbers5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stanumbers5 + END) AS most_common_freqs, + (CASE + WHEN (stat.a).stakind1 = 2 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 2 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 2 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 2 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 2 THEN (stat.a).stavalues5 + END) AS histogram_bounds, + (CASE + WHEN (stat.a).stakind1 = 3 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 3 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 3 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 3 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 3 THEN (stat.a).stanumbers5[1] + END) correlation, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx + +-- Specify most_common_vals_kind as 8-STATISTIC_KIND_MCV_VALUE_SORTED. +SELECT pg_catalog.pg_restore_extended_stats( + 'schemaname', 'stats_import', + 'relname', 'test_clone', + 'statistics_schemaname', 'stats_import', + 'statistics_name', 'test_stat_clone', + 'inherited', false, + 'exprs', '[ + { "most_common_vals": "{11,12,13}", "most_common_freqs": "{0.11,0.12,0.13}", "most_common_vals_kind":"8" }, + { "most_common_vals": "{41,42,43}", "most_common_freqs": "{0.011,0.022,0.033}","most_common_vals_kind":"8" } + ]'::jsonb); + +SELECT cn.nspname AS schemaname, + c.relname AS tablename, + s.stxrelid AS tableid, + sn.nspname AS statistics_schemaname, + s.stxname AS statistics_name, + s.oid AS statistics_id, + pg_get_userbyid(s.stxowner) AS statistics_owner, + stat.expr, + sd.stxdinherit AS inherited, + (stat.a).stanullfrac AS null_frac, + (stat.a).stawidth AS avg_width, + (stat.a).stadistinct AS n_distinct, + (CASE + WHEN (stat.a).stakind1 = 1 THEN 1 + WHEN (stat.a).stakind2 = 1 THEN 1 + WHEN (stat.a).stakind3 = 1 THEN 1 + WHEN (stat.a).stakind4 = 1 THEN 1 + WHEN (stat.a).stakind5 = 1 THEN 1 + WHEN (stat.a).stakind1 = 8 THEN 8 + WHEN (stat.a).stakind2 = 8 THEN 8 + WHEN (stat.a).stakind3 = 8 THEN 8 + WHEN (stat.a).stakind4 = 8 THEN 8 + WHEN (stat.a).stakind5 = 8 THEN 8 + else -1 + END) AS most_common_vals_kind, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stavalues5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stavalues5 + END) AS most_common_vals, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stanumbers5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stanumbers5 + END) AS most_common_freqs, + (CASE + WHEN (stat.a).stakind1 = 2 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 2 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 2 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 2 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 2 THEN (stat.a).stavalues5 + END) AS histogram_bounds, + (CASE + WHEN (stat.a).stakind1 = 3 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 3 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 3 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 3 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 3 THEN (stat.a).stanumbers5[1] + END) correlation, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx + +DROP SCHEMA stats_import CASCADE; + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx + +-- Specify most_common_vals_kind as 8-STATISTIC_KIND_MCV_VALUE_SORTED. +SELECT pg_catalog.pg_restore_extended_stats( + 'schemaname', 'stats_import', + 'relname', 'test_clone', + 'statistics_schemaname', 'stats_import', + 'statistics_name', 'test_stat_clone', + 'inherited', false, + 'exprs', '[ + { "most_common_vals": "{11,12,13}", "most_common_freqs": "{0.11,0.12,0.13}", "most_common_vals_kind":"8" }, + { "most_common_vals": "{41,42,43}", "most_common_freqs": "{0.011,0.022,0.033}","most_common_vals_kind":"8" } + ]'::jsonb); + +SELECT cn.nspname AS schemaname, + c.relname AS tablename, + s.stxrelid AS tableid, + sn.nspname AS statistics_schemaname, + s.stxname AS statistics_name, + s.oid AS statistics_id, + pg_get_userbyid(s.stxowner) AS statistics_owner, + stat.expr, + sd.stxdinherit AS inherited, + (stat.a).stanullfrac AS null_frac, + (stat.a).stawidth AS avg_width, + (stat.a).stadistinct AS n_distinct, + (CASE + WHEN (stat.a).stakind1 = 1 THEN 1 + WHEN (stat.a).stakind2 = 1 THEN 1 + WHEN (stat.a).stakind3 = 1 THEN 1 + WHEN (stat.a).stakind4 = 1 THEN 1 + WHEN (stat.a).stakind5 = 1 THEN 1 + WHEN (stat.a).stakind1 = 8 THEN 8 + WHEN (stat.a).stakind2 = 8 THEN 8 + WHEN (stat.a).stakind3 = 8 THEN 8 + WHEN (stat.a).stakind4 = 8 THEN 8 + WHEN (stat.a).stakind5 = 8 THEN 8 + else -1 + END) AS most_common_vals_kind, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stavalues5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stavalues5 + END) AS most_common_vals, + (CASE + WHEN (stat.a).stakind1 = 1 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 1 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 1 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 1 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 1 THEN (stat.a).stanumbers5 + WHEN (stat.a).stakind1 = 8 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 8 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 8 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 8 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 8 THEN (stat.a).stanumbers5 + END) AS most_common_freqs, + (CASE + WHEN (stat.a).stakind1 = 2 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 2 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 2 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 2 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 2 THEN (stat.a).stavalues5 + END) AS histogram_bounds, + (CASE + WHEN (stat.a).stakind1 = 3 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 3 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 3 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 3 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 3 THEN (stat.a).stanumbers5[1] + END) correlation, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stavalues5 + END) AS most_common_elems, + (CASE + WHEN (stat.a).stakind1 = 4 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 4 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 4 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 4 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 4 THEN (stat.a).stanumbers5 + END) AS most_common_elem_freqs, + (CASE + WHEN (stat.a).stakind1 = 5 THEN (stat.a).stanumbers1 + WHEN (stat.a).stakind2 = 5 THEN (stat.a).stanumbers2 + WHEN (stat.a).stakind3 = 5 THEN (stat.a).stanumbers3 + WHEN (stat.a).stakind4 = 5 THEN (stat.a).stanumbers4 + WHEN (stat.a).stakind5 = 5 THEN (stat.a).stanumbers5 + END) AS elem_count_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stavalues5 + END) AS range_length_histogram, + (CASE + WHEN (stat.a).stakind1 = 6 THEN (stat.a).stanumbers1[1] + WHEN (stat.a).stakind2 = 6 THEN (stat.a).stanumbers2[1] + WHEN (stat.a).stakind3 = 6 THEN (stat.a).stanumbers3[1] + WHEN (stat.a).stakind4 = 6 THEN (stat.a).stanumbers4[1] + WHEN (stat.a).stakind5 = 6 THEN (stat.a).stanumbers5[1] + END) AS range_empty_frac, + (CASE + WHEN (stat.a).stakind1 = 7 THEN (stat.a).stavalues1 + WHEN (stat.a).stakind2 = 7 THEN (stat.a).stavalues2 + WHEN (stat.a).stakind3 = 7 THEN (stat.a).stavalues3 + WHEN (stat.a).stakind4 = 7 THEN (stat.a).stavalues4 + WHEN (stat.a).stakind5 = 7 THEN (stat.a).stavalues5 + END) AS range_bounds_histogram + FROM pg_statistic_ext s JOIN pg_class c ON (c.oid = s.stxrelid) + LEFT JOIN pg_statistic_ext_data sd ON (s.oid = sd.stxoid) + LEFT JOIN pg_namespace cn ON (cn.oid = c.relnamespace) + LEFT JOIN pg_namespace sn ON (sn.oid = s.stxnamespace) + JOIN LATERAL ( + SELECT unnest(pg_get_statisticsobjdef_expressions(s.oid)) AS expr, + unnest(sd.stxdexpr)::pg_statistic AS a + ) stat ON (stat.expr IS NOT NULL) + WHERE pg_has_role(c.relowner, 'USAGE') + AND (c.relrowsecurity = false OR NOT row_security_active(c.oid)) + AND sn.nspname = 'stats_import' AND s.stxname = 'test_stat_clone'\gx + DROP SCHEMA stats_import CASCADE; -- 2.43.0