From 06dafe47952b13cf5be181866637d78be86bf194 Mon Sep 17 00:00:00 2001 From: Zsolt Parragi Date: Mon, 20 Jul 2026 10:12:47 +0000 Subject: [PATCH v2 6/8] Add JSON5 string escapes and multi-line strings to the JSON lexer When json5 is set, strings follow the ECMAScript 5.1 escape rules: besides the JSON escapes, accept \', \v, \0 and \xHH, and let any other non-digit character escape to itself. A backslash before a line terminator (LF, CR, CRLF, U+2028, U+2029) continues the string on the next line, advancing line tracking. Raw control characters other than line breaks are allowed inside strings. The strict JSON escape paths are unchanged. --- src/common/jsonapi.c | 140 ++++++++++++++++-- .../test_json_parser/json5_multiline.json5 | 9 ++ .../test_json_parser/json5_multiline.out | 5 + .../test_json_parser/t/005_test_json5.pl | 110 ++++++++++++-- 4 files changed, 242 insertions(+), 22 deletions(-) create mode 100644 src/test/modules/test_json_parser/json5_multiline.json5 create mode 100644 src/test/modules/test_json_parser/json5_multiline.out diff --git a/src/common/jsonapi.c b/src/common/jsonapi.c index 8dbc66512fc..e418847c82f 100644 --- a/src/common/jsonapi.c +++ b/src/common/jsonapi.c @@ -2609,6 +2609,116 @@ json_lex(JsonLexContext *lex) return JSON_SUCCESS; } +/* + * Lex a json5 escape sequence that JSON doesn't have, for json_lex_string(). + * + * json5 escapes follow ECMAScript 5.1. Besides the JSON ones, there are + * \', \v, \0 and \xHH; a backslash before a line terminator continues the + * string on the next line; and any other character except a digit stands + * for itself. If that is a multibyte character, only its first byte is + * consumed here, the caller copies the rest as ordinary string content. + * + * s points at the character after the backslash. Returns the position of + * the last character of the sequence, or NULL after setting *err and the + * error position in lex. This is kept out of line so that it doesn't weigh + * on the strict JSON string lexing it is inlined into, and returns the + * position so that the caller's cursor can stay in a register. + */ +static pg_noinline const char * +json5_lex_escape(JsonLexContext *lex, const char *s, JsonParseErrorType *err) +{ + const char *const end = lex->input + lex->input_length; + int lslen = 0; + + /* like json_lex_string()'s FAIL_AT_CHAR_END */ +#define FAIL_AT_CHAR_END(code) \ + do { \ + ptrdiff_t remaining = end - s; \ + int charlen; \ + charlen = pg_encoding_mblen_or_incomplete(lex->input_encoding, \ + s, remaining); \ + lex->token_terminator = (charlen <= remaining) ? s + charlen : end; \ + *err = (code); \ + return NULL; \ + } while (0) + + if (*s == '\n' || *s == '\r' || + (lslen = json5_line_separator_len(lex, s, end)) > 0) + { + if (*s == '\r' && s + 1 < end && *(s + 1) == '\n') + s++; + if (*s == '\n' || *s == '\r') + { + ++lex->line_number; + lex->line_start = s + 1; + } + else + s += lslen - 1; + } + else if (*s == 'x') + { + const char *x = s; + char32_t ch = 0; + int i; + + for (i = 1; i <= 2; i++) + { + s++; + if (s >= end) + { + /* json5 is never lexed incrementally */ + lex->token_terminator = s; + *err = JSON_INVALID_TOKEN; + return NULL; + } + else if (*s >= '0' && *s <= '9') + ch = (ch * 16) + (*s - '0'); + else if (*s >= 'a' && *s <= 'f') + ch = (ch * 16) + (*s - 'a') + 10; + else if (*s >= 'A' && *s <= 'F') + ch = (ch * 16) + (*s - 'A') + 10; + else + { + lex->token_start = x; + FAIL_AT_CHAR_END(JSON_ESCAPING_INVALID); + } + } + if (lex->need_escapes) + { + JsonParseErrorType result; + + /* same restriction as for \u0000 */ + if (ch == 0) + FAIL_AT_CHAR_END(JSON_UNICODE_CODE_POINT_ZERO); + result = append_codepoint(lex, ch); + if (result != JSON_SUCCESS) + FAIL_AT_CHAR_END(result); + } + } + else if (*s == '0' && !(s + 1 < end && *(s + 1) >= '0' && *(s + 1) <= '9')) + { + if (lex->need_escapes) + FAIL_AT_CHAR_END(JSON_UNICODE_CODE_POINT_ZERO); + } + else if ((*s >= '0' && *s <= '9') || *s == '\0') + { + /* octal-looking escapes aren't allowed, nor is a raw NUL */ + lex->token_start = s; + FAIL_AT_CHAR_END(JSON_ESCAPING_INVALID); + } + else if (lex->need_escapes) + { + if (*s == 'v') + jsonapi_appendStringInfoChar(lex->strval, '\v'); + else + jsonapi_appendStringInfoChar(lex->strval, *s); + } + + return s; + +#undef FAIL_AT_CHAR_END +} + /* * The next token in the input stream is known to be a string; lex it. * @@ -2737,6 +2847,16 @@ json_lex_string(JsonLexContext *lex) FAIL_AT_CHAR_END(result); } } + else if (unlikely(lex->json5) && + (*s == '\0' || strchr("\"\\/bfnrt", *s) == NULL)) + { + /* json5-only escape; the JSON ones are handled below */ + if (hi_surrogate != -1) + FAIL_AT_CHAR_END(JSON_UNICODE_LOW_SURROGATE); + s = json5_lex_escape(lex, s, &result); + if (s == NULL) + return result; + } else if (lex->need_escapes) { if (hi_surrogate != -1) @@ -2749,14 +2869,6 @@ json_lex_string(JsonLexContext *lex) case '/': jsonapi_appendStringInfoChar(lex->strval, *s); break; - case '\'': - if (!lex->json5) - { - lex->token_start = s; - FAIL_AT_CHAR_END(JSON_ESCAPING_INVALID); - } - jsonapi_appendStringInfoChar(lex->strval, *s); - break; case 'b': jsonapi_appendStringInfoChar(lex->strval, '\b'); break; @@ -2783,8 +2895,7 @@ json_lex_string(JsonLexContext *lex) FAIL_AT_CHAR_END(JSON_ESCAPING_INVALID); } } - else if (strchr("\"\\/bfnrt", *s) == NULL && - !(lex->json5 && *s == '\'')) + else if (strchr("\"\\/bfnrt", *s) == NULL) { /* * Simpler processing if we're not bothered about de-escaping @@ -2820,7 +2931,16 @@ json_lex_string(JsonLexContext *lex) break; else if ((unsigned char) *p <= 31) { + /* + * json5 allows raw control characters other than line + * breaks. NUL is still rejected, as a token can't hold + * it. + */ + if (lex->json5 && *p != '\n' && *p != '\r' && *p != '\0') + continue; + /* Per RFC4627, these characters MUST be escaped. */ + /* * Since *p isn't printable, exclude it from the context * string diff --git a/src/test/modules/test_json_parser/json5_multiline.json5 b/src/test/modules/test_json_parser/json5_multiline.json5 new file mode 100644 index 00000000000..7a4050c5d75 --- /dev/null +++ b/src/test/modules/test_json_parser/json5_multiline.json5 @@ -0,0 +1,9 @@ +{ + a: 'line one \ +line two \ +line three', + b: "double one \ +double two", + c: 'crlf one \ +crlf two' +} diff --git a/src/test/modules/test_json_parser/json5_multiline.out b/src/test/modules/test_json_parser/json5_multiline.out new file mode 100644 index 00000000000..3a0ba5c2fe7 --- /dev/null +++ b/src/test/modules/test_json_parser/json5_multiline.out @@ -0,0 +1,5 @@ +{ +"a": "line one line two line three", +"b": "double one double two", +"c": "crlf one crlf two" +} diff --git a/src/test/modules/test_json_parser/t/005_test_json5.pl b/src/test/modules/test_json_parser/t/005_test_json5.pl index 847b1c53942..972a1fdb62f 100644 --- a/src/test/modules/test_json_parser/t/005_test_json5.pl +++ b/src/test/modules/test_json_parser/t/005_test_json5.pl @@ -39,34 +39,82 @@ my @json5_inline_cases = ( [ 'line comment only', "// c", undef ], [ 'line comment ended by U+2028', - "[1, // c\xe2\x80\xa8 2]", "[\n1,\n2\n]" + "[1, // c\xe2\x80\xa8 2]", + "[\n1,\n2\n]" ], [ 'line comment ended by U+2029', - "[1, // c\xe2\x80\xa9 2]", "[\n1,\n2\n]" + "[1, // c\xe2\x80\xa9 2]", + "[\n1,\n2\n]" ], [ 'non-space symbol outside strings', "[1,\xe2\x80\xa6 2]", undef ], [ 'invalid UTF-8 outside strings', "[1,\xc2 2]", undef ], [ 'no-break space around unquoted key', - "{\xc2\xa0a\xc2\xa0: 1}", "{\n\"a\": 1\n}" + "{\xc2\xa0a\xc2\xa0: 1}", + "{\n\"a\": 1\n}" + ], + [ + 'non-ASCII letter in key', + "{\xc3\xa9t\xc3\xa9: 1}", + "{\n\"\xc3\xa9t\xc3\xa9\": 1\n}" + ], + [ + 'combining mark after first key char', + "{a\xcc\x81: 1}", + "{\n\"a\xcc\x81\": 1\n}" ], - [ 'non-ASCII letter in key', "{\xc3\xa9t\xc3\xa9: 1}", "{\n\"\xc3\xa9t\xc3\xa9\": 1\n}" ], - [ 'combining mark after first key char', "{a\xcc\x81: 1}", "{\n\"a\xcc\x81\": 1\n}" ], [ 'combining mark as first key char', "{\xcc\x81a: 1}", undef ], - [ 'zero width non-joiner in key', "{a\xe2\x80\x8cb: 1}", "{\n\"a\xe2\x80\x8cb\": 1\n}" ], + [ + 'zero width non-joiner in key', + "{a\xe2\x80\x8cb: 1}", + "{\n\"a\xe2\x80\x8cb\": 1\n}" + ], [ 'non-identifier symbol in key', "{a\xe2\x80\xa6b: 1}", undef ], [ 'escaped key characters', "{\\u0061b\\u0063: 1}", "{\n\"abc\": 1\n}" ], - [ 'escaped non-ASCII key character', "{\\u00e9: 1}", "{\n\"\xc3\xa9\": 1\n}" ], + [ + 'escaped non-ASCII key character', + "{\\u00e9: 1}", + "{\n\"\xc3\xa9\": 1\n}" + ], [ 'escaped space in key', "{a\\u0020b: 1}", undef ], [ 'malformed escape in key', "{a\\u00zz: 1}", undef ], [ 'escaped surrogate pair in key', "{\\ud83d\\ude00: 1}", undef ], - [ 'keyword continued by escape', "{true\\u0061: 1}", "{\n\"truea\": 1\n}" ], + [ + 'keyword continued by escape', + "{true\\u0061: 1}", + "{\n\"truea\": 1\n}" + ], [ 'escaped keyword as key', "{\\u0074rue: 1}", "{\n\"true\": 1\n}" ], [ 'escaped keyword as value', "[\\u0074rue]", undef ], [ 'keyword followed by no-break space', "[true\xc2\xa0]", "[\ntrue\n]" ], - [ 'Infinity followed by no-break space', "[Infinity\xc2\xa0]", "[\nInfinity\n]" ], - [ 'NaN key followed by no-break space', "{NaN\xc2\xa0: 1}", "{\n\"NaN\": 1\n}" ],); + [ + 'Infinity followed by no-break space', "[Infinity\xc2\xa0]", + "[\nInfinity\n]" + ], + [ + 'NaN key followed by no-break space', + "{NaN\xc2\xa0: 1}", + "{\n\"NaN\": 1\n}" + ], + [ 'vertical tab escape', "['\\v']", "[\n\"\\u000b\"\n]" ], + [ 'hex escapes', "['\\x41\\x6a\\xe9']", "[\n\"Aj\xc3\xa9\"\n]" ], + [ 'malformed hex escape', "['\\x4g']", undef ], + [ 'truncated hex escape', "['\\x4", undef ], + [ 'zero escape before digit', "['\\01']", undef ], + [ 'decimal digit escape', "['\\1']", undef ], + [ 'escaped non-escape characters', "['\\a\\q\\\"']", "[\n\"aq\\\"\"\n]" ], + [ 'escaped non-ASCII character', "['\\\xc3\xa9']", "[\n\"\xc3\xa9\"\n]" ], + [ + 'line continuation with U+2028', "['a\\\xe2\x80\xa8b']", + "[\n\"ab\"\n]" + ], + [ 'line continuation with CRLF', "['a\\\r\nb']", "[\n\"ab\"\n]" ], + [ 'raw tab in string', "['a\tb']", "[\n\"a\\tb\"\n]" ], + [ 'raw newline in string', "['a\nb']", undef ], + [ 'raw NUL in string', "['a\0b']", undef ], + [ 'escaped raw NUL in string', "['a\\\0b']", undef ], + [ 'high surrogate before hex escape', "['\\ud83d\\x41']", undef ],); # One entry per feature: fixture basename plus, where the failing token # is predictable, a regex for the error reported without --json5. @@ -79,6 +127,7 @@ my @features = ( { name => 'trailing commas', file => 'json5_trailing_commas' }, { name => 'unquoted keys', file => 'json5_keys' }, { name => 'single-quoted strings', file => 'json5_strings' }, + { name => 'multi-line strings', file => 'json5_multiline' }, { name => 'numbers', file => 'json5_numbers' },); # Inputs that stay invalid even in json5 mode. @@ -232,8 +281,8 @@ foreach my $exe (@exes) } else { - check_rejected($exe, $fname, "json5 mode: $label", undef, - "--json5"); + check_rejected($exe, $fname, "json5 mode: $label", + undef, "--json5"); } } @@ -273,6 +322,43 @@ foreach my $exe (@exes) check_rejected($exe, inline_file($content), "non-json5 mode: $label form"); } + + # Multi-line string continuations are also handled on the "not + # de-escaping" path (no -s flag). + my $ml_file = "$FindBin::RealBin/../json5_multiline.json5"; + my ($ml_out, $ml_err) = run_parser($exe, "--json5", $ml_file); + + like($ml_out, qr/SUCCESS/, + "json5 multi-line strings, no de-escaping: parse succeeds"); + is($ml_err, "", + "json5 multi-line strings, no de-escaping: no error output"); + + # A backslash-newline continuation in a quoted key's value exercises + # the gate directly: a quoted key rules out the unrelated + # unquoted-key rejection reached via the fixture file. + my $cont_file = inline_file("{ \"a\": \"line \\\nb\" }"); + my ($cont_out, $cont_err) = + run_parser($exe, "-s", "--json5", $cont_file); + + is( $cont_out, + "{\n\"a\": \"line b\"\n}", + "json5 mode: backslash-newline continuation accepted"); + is($cont_err, "", "json5 mode: backslash-newline continuation no error"); + + check_rejected( + $exe, $cont_file, + "non-json5 mode: backslash-newline continuation", + qr/Escape sequence.*is invalid/s); + + # Same continuation with a lone CR (no LF) as the line terminator. + my $cr_file = inline_file("{ \"a\": \"line \\\rb\" }"); + my ($cr_out, $cr_err) = + run_parser($exe, "-s", "--json5", $cr_file); + + is( $cr_out, + "{\n\"a\": \"line b\"\n}", + "json5 mode: backslash-cr continuation accepted"); + is($cr_err, "", "json5 mode: backslash-cr continuation no error"); } done_testing(); -- 2.55.0