From 431eb866175c483d00bb0aa41467b98c8a175eab Mon Sep 17 00:00:00 2001 From: Rui Zhao Date: Sun, 4 Oct 2026 23:14:02 +0800 Subject: [PATCH 2/3] Avoid allocating a C string for the normalization form The ASCII fast path still copies the normalization form into a C string on every call. Compare the short form names directly in the text value, and allocate a C string only for reporting an invalid form. --- src/backend/utils/adt/varlena.c | 22 ++++++++++++---------- src/test/regress/expected/unicode.out | 10 ++++++++++ src/test/regress/sql/unicode.sql | 3 +++ 3 files changed, 25 insertions(+), 10 deletions(-) diff --git a/src/backend/utils/adt/varlena.c b/src/backend/utils/adt/varlena.c index 9161abc9361..8186687debe 100644 --- a/src/backend/utils/adt/varlena.c +++ b/src/backend/utils/adt/varlena.c @@ -5421,8 +5421,10 @@ getClosestMatch(ClosestMatchState *state) */ static UnicodeNormalizationForm -unicode_norm_form_from_string(const char *formstr) +unicode_norm_form_from_text(text *formtxt) { + const char *formstr = VARDATA_ANY(formtxt); + int formlen = VARSIZE_ANY_EXHDR(formtxt); UnicodeNormalizationForm form = -1; /* @@ -5433,18 +5435,18 @@ unicode_norm_form_from_string(const char *formstr) (errcode(ERRCODE_SYNTAX_ERROR), errmsg("Unicode normalization can only be performed if server encoding is UTF8"))); - if (pg_strcasecmp(formstr, "NFC") == 0) + if (formlen == 3 && pg_strncasecmp(formstr, "NFC", 3) == 0) form = UNICODE_NFC; - else if (pg_strcasecmp(formstr, "NFD") == 0) + else if (formlen == 3 && pg_strncasecmp(formstr, "NFD", 3) == 0) form = UNICODE_NFD; - else if (pg_strcasecmp(formstr, "NFKC") == 0) + else if (formlen == 4 && pg_strncasecmp(formstr, "NFKC", 4) == 0) form = UNICODE_NFKC; - else if (pg_strcasecmp(formstr, "NFKD") == 0) + else if (formlen == 4 && pg_strncasecmp(formstr, "NFKD", 4) == 0) form = UNICODE_NFKD; else ereport(ERROR, (errcode(ERRCODE_INVALID_PARAMETER_VALUE), - errmsg("invalid normalization form: %s", formstr))); + errmsg("invalid normalization form: %s", text_to_cstring(formtxt)))); return form; } @@ -5548,7 +5550,7 @@ Datum unicode_normalize_func(PG_FUNCTION_ARGS) { text *input = PG_GETARG_TEXT_PP(0); - char *formstr = text_to_cstring(PG_GETARG_TEXT_PP(1)); + text *formtxt = PG_GETARG_TEXT_PP(1); UnicodeNormalizationForm form; size_t size; char32_t *input_chars; @@ -5559,7 +5561,7 @@ unicode_normalize_func(PG_FUNCTION_ARGS) int len; int start; - form = unicode_norm_form_from_string(formstr); + form = unicode_norm_form_from_text(formtxt); /* * ASCII characters have no decomposition and a combining class of zero, @@ -5632,7 +5634,7 @@ Datum unicode_is_normalized(PG_FUNCTION_ARGS) { text *input = PG_GETARG_TEXT_PP(0); - char *formstr = text_to_cstring(PG_GETARG_TEXT_PP(1)); + text *formtxt = PG_GETARG_TEXT_PP(1); UnicodeNormalizationForm form; size_t size; char32_t *input_chars; @@ -5645,7 +5647,7 @@ unicode_is_normalized(PG_FUNCTION_ARGS) int len; int start; - form = unicode_norm_form_from_string(formstr); + form = unicode_norm_form_from_text(formtxt); /* * Leading ASCII is unchanged by normalization, except that its last diff --git a/src/test/regress/expected/unicode.out b/src/test/regress/expected/unicode.out index 655dd99750a..ea93ad9c945 100644 --- a/src/test/regress/expected/unicode.out +++ b/src/test/regress/expected/unicode.out @@ -89,6 +89,14 @@ FROM (VALUES ('abc'::text)) v(val); SELECT "normalize"('abc', 'def'); -- run-time error ERROR: invalid normalization form: def +SELECT "normalize"('abc', 'NFCx'); -- run-time error +ERROR: invalid normalization form: NFCx +SELECT "normalize"(U&'\0061\0308', 'nFc') = U&'\00E4' COLLATE "C" AS test_form_case; + test_form_case +---------------- + t +(1 row) + SELECT U&'\00E4\24D1c' IS NORMALIZED AS test_default; test_default -------------- @@ -130,6 +138,8 @@ ORDER BY num; SELECT is_normalized('abc', 'def'); -- run-time error ERROR: invalid normalization form: def +SELECT is_normalized('abc', 'NFKCx'); -- run-time error +ERROR: invalid normalization form: NFKCx -- Exercise the ASCII fast-path's chunk/remainder boundary handling: a -- non-NFC-normalized codepoint sequence must still be detected as such -- regardless of how much pure-ASCII padding surrounds it. diff --git a/src/test/regress/sql/unicode.sql b/src/test/regress/sql/unicode.sql index 373745b32d3..81e08d99b81 100644 --- a/src/test/regress/sql/unicode.sql +++ b/src/test/regress/sql/unicode.sql @@ -22,6 +22,8 @@ SELECT normalize(val) = val COLLATE "C" AS test_ascii_idem_column FROM (VALUES ('abc'::text)) v(val); SELECT "normalize"('abc', 'def'); -- run-time error +SELECT "normalize"('abc', 'NFCx'); -- run-time error +SELECT "normalize"(U&'\0061\0308', 'nFc') = U&'\00E4' COLLATE "C" AS test_form_case; SELECT U&'\00E4\24D1c' IS NORMALIZED AS test_default; SELECT U&'\00E4\24D1c' IS NFC NORMALIZED AS test_nfc; @@ -41,6 +43,7 @@ FROM ORDER BY num; SELECT is_normalized('abc', 'def'); -- run-time error +SELECT is_normalized('abc', 'NFKCx'); -- run-time error -- Exercise the ASCII fast-path's chunk/remainder boundary handling: a -- non-NFC-normalized codepoint sequence must still be detected as such -- 2.43.7