From 9b357d2502cb90104bdd7622a2d8e50ad61fa5eb Mon Sep 17 00:00:00 2001 From: SungJun Jang Date: Thu, 1 Oct 2026 11:20:09 +0900 Subject: [PATCH v4 2/2] Make EUC-KR encoding routines self-contained Per KS X 2901 (formerly KS C 5861-1992), EUC-KR designates only G0 (ASCII) and G1 (KS X 1001). G2 and G3 are not designated, so SS2 (0x8E) and SS3 (0x8F) cannot appear as lead bytes and no 3-byte sequence is ever valid in EUC-KR. pg_euckr_verifychar() already reflects this: it has no SS2/SS3 case. But pg_euckr_mblen(), pg_euckr_dsplen(), and pg_euckr2wchar_with_len() delegated to the shared pg_euc_* helpers, which include SS2/SS3 handling for encodings that designate G2/G3. Replace these with EUC-KR-specific implementations, change pg_wchar_table[PG_EUC_KR] maxmblen from 3 to 2, and update the "Bytes/Char" column of the character set table in the documentation from "1-3" to "1-2". After this patch, EUC-KR's mb routines are structurally identical to UHC's: IS_HIGHBIT_SET-only dispatch, 1-2 byte sequences, maxmblen=2. Valid EUC-KR data is handled exactly as before, since the verifier never admits SS2 or SS3 as lead bytes. The user-visible effects are: * pg_encoding_max_length('EUC_KR') returns 2 instead of 3. * In EUC_KR databases, the maximum size computed for char(n) and varchar(n) shrinks from 3 to 2 bytes per character. The planner's width estimates for such columns shrink accordingly, for example from 34 to 24 for char(10), which can change row and cost estimates and therefore query plans. For the same reason, a new table whose columns all have a bounded width might no longer get a TOAST table, for example a table with a single varchar(700) column. * 0x8F is now taken as the lead byte of a 2-byte sequence rather than a 3-byte one; 0x8E was already taken as 2 bytes. Error messages about an invalid byte sequence starting with 0x8F therefore show two bytes instead of three. Text that contains 0x8F and was stored without being verified as EUC_KR, such as a shared catalog entry written from a database using another encoding, is split into characters differently. Author: SungJun Jang Reviewed-by: Henson Choi Reviewed-by: Michael Paquier Reviewed-by: John Naylor Discussion: https://postgr.es/m/CAE+cgNgTWvCT2+HZYRzQA8-wSQrj-FjPQQNffn=_3DOpz0pKgA@mail.gmail.com --- doc/src/sgml/charset.sgml | 2 +- src/common/wchar.c | 36 ++++++++++++++++++++++++---- src/test/regress/expected/euc_kr.out | 4 ++-- 3 files changed, 35 insertions(+), 7 deletions(-) diff --git a/doc/src/sgml/charset.sgml b/doc/src/sgml/charset.sgml index 89760466c1f..a283c9c41ae 100644 --- a/doc/src/sgml/charset.sgml +++ b/doc/src/sgml/charset.sgml @@ -1862,7 +1862,7 @@ ORDER BY c COLLATE ebcdic; Korean Yes Yes - 1–3 + 1–2 diff --git a/src/common/wchar.c b/src/common/wchar.c index 926823cabec..c87c41ec3ca 100644 --- a/src/common/wchar.c +++ b/src/common/wchar.c @@ -210,23 +210,51 @@ pg_eucjp_dsplen(const unsigned char *s) /* * EUC_KR + * + * Per KS X 2901 (formerly KS C 5861-1992), EUC-KR designates only G0 + * (ASCII) and G1 (KS X 1001). G2 and G3 are not designated, so the + * single-shift codes SS2 (0x8E) and SS3 (0x8F) never appear as lead + * bytes and no 3-byte sequence is ever valid. These routines therefore + * implement EUC-KR directly rather than delegating to the shared + * pg_euc_* helpers, which include SS2/SS3 handling for encodings that + * designate G2/G3. */ static int pg_euckr2wchar_with_len(const unsigned char *from, pg_wchar *to, int len) { - return pg_euc2wchar_with_len(from, to, len); + int cnt = 0; + + while (len > 0 && *from) + { + if (IS_HIGHBIT_SET(*from)) /* G1: KS X 1001, 2 bytes */ + { + MB2CHAR_NEED_AT_LEAST(len, 2); + *to = *from++ << 8; + *to |= *from++; + len -= 2; + } + else /* G0: ASCII */ + { + *to = *from++; + len--; + } + to++; + cnt++; + } + *to = 0; + return cnt; } static int pg_euckr_mblen(const unsigned char *s) { - return pg_euc_mblen(s); + return IS_HIGHBIT_SET(*s) ? 2 : 1; } static int pg_euckr_dsplen(const unsigned char *s) { - return pg_euc_dsplen(s); + return IS_HIGHBIT_SET(*s) ? 2 : pg_ascii_dsplen(s); } /* @@ -1866,7 +1894,7 @@ const pg_wchar_tbl pg_wchar_table[] = { [PG_SQL_ASCII] = {pg_ascii2wchar_with_len, pg_wchar2single_with_len, pg_ascii_mblen, pg_ascii_dsplen, pg_ascii_verifychar, pg_ascii_verifystr, 1}, [PG_EUC_JP] = {pg_eucjp2wchar_with_len, pg_wchar2euc_with_len, pg_eucjp_mblen, pg_eucjp_dsplen, pg_eucjp_verifychar, pg_eucjp_verifystr, 3}, [PG_EUC_CN] = {pg_euccn2wchar_with_len, pg_wchar2euc_with_len, pg_euccn_mblen, pg_euccn_dsplen, pg_euccn_verifychar, pg_euccn_verifystr, 3}, - [PG_EUC_KR] = {pg_euckr2wchar_with_len, pg_wchar2euc_with_len, pg_euckr_mblen, pg_euckr_dsplen, pg_euckr_verifychar, pg_euckr_verifystr, 3}, + [PG_EUC_KR] = {pg_euckr2wchar_with_len, pg_wchar2euc_with_len, pg_euckr_mblen, pg_euckr_dsplen, pg_euckr_verifychar, pg_euckr_verifystr, 2}, [PG_EUC_TW] = {pg_euctw2wchar_with_len, pg_wchar2euc_with_len, pg_euctw_mblen, pg_euctw_dsplen, pg_euctw_verifychar, pg_euctw_verifystr, 4}, [PG_EUC_JIS_2004] = {pg_eucjp2wchar_with_len, pg_wchar2euc_with_len, pg_eucjp_mblen, pg_eucjp_dsplen, pg_eucjp_verifychar, pg_eucjp_verifystr, 3}, [PG_UTF8] = {pg_utf2wchar_with_len, pg_wchar2utf_with_len, pg_utf_mblen, pg_utf_dsplen, pg_utf8_verifychar, pg_utf8_verifystr, 4}, diff --git a/src/test/regress/expected/euc_kr.out b/src/test/regress/expected/euc_kr.out index 9ddf9f85cb0..2683a672a37 100644 --- a/src/test/regress/expected/euc_kr.out +++ b/src/test/regress/expected/euc_kr.out @@ -18,7 +18,7 @@ SELECT POSITION( SELECT pg_encoding_max_length(pg_char_to_encoding('EUC_KR')); pg_encoding_max_length ------------------------ - 3 + 2 (1 row) -- Code set 0 (ASCII) characters take one byte and code set 1 (KS X 1001) @@ -48,4 +48,4 @@ SELECT regexp_replace(convert_from('\x41b0fac7d0', 'EUC_KR'), '[^A]', 'x', 'g'); SELECT convert_from('\x8ea1', 'EUC_KR'); ERROR: invalid byte sequence for encoding "EUC_KR": 0x8e 0xa1 SELECT convert_from('\x8fa1a1', 'EUC_KR'); -ERROR: invalid byte sequence for encoding "EUC_KR": 0x8f 0xa1 0xa1 +ERROR: invalid byte sequence for encoding "EUC_KR": 0x8f 0xa1 -- 2.48.1.windows.1