From da801156c02c42617df63f9b6b843325c9ecc47a Mon Sep 17 00:00:00 2001 From: Rui Zhao Date: Sun, 4 Oct 2026 23:12:59 +0800 Subject: [PATCH 1/3] Avoid counting code points before checking Unicode assignment unicode_assigned() already knows the byte boundary of the remaining input. Walk to that boundary instead of calling pg_mbstrlen_with_len() first, avoiding a second traversal and allowing an unassigned code point to stop the scan without reading the remaining suffix. --- src/backend/utils/adt/varlena.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/src/backend/utils/adt/varlena.c b/src/backend/utils/adt/varlena.c index 11eb212384d..9161abc9361 100644 --- a/src/backend/utils/adt/varlena.c +++ b/src/backend/utils/adt/varlena.c @@ -5513,8 +5513,8 @@ unicode_assigned(PG_FUNCTION_ARGS) { text *input = PG_GETARG_TEXT_PP(0); unsigned char *p; + unsigned char *end; int len; - int size; int start; if (GetDatabaseEncoding() != PG_UTF8) @@ -5527,10 +5527,10 @@ unicode_assigned(PG_FUNCTION_ARGS) if (start == len) PG_RETURN_BOOL(true); - /* convert to char32_t */ - size = pg_mbstrlen_with_len(VARDATA_ANY(input) + start, len - start); + /* Check the remaining code points without first counting them. */ p = (unsigned char *) VARDATA_ANY(input) + start; - for (int i = 0; i < size; i++) + end = (unsigned char *) VARDATA_ANY(input) + len; + while (p < end && *p) { char32_t uchar = utf8_to_unicode(p); int category = unicode_category(uchar); -- 2.43.7