From 8705a038b6bf945d0bbd497e017387d6d3e9c24f Mon Sep 17 00:00:00 2001 From: Zsolt Parragi Date: Mon, 20 Jul 2026 09:00:25 +0000 Subject: [PATCH v2 1/8] Add JSON5 comment and whitespace support to the JSON lexer Introduce a json5 flag on JsonLexContext (plus a param on makeJsonLexContextCstringLen, all existing callers pass false) and, when set, skip // line and /* */ block comments wherever whitespace is allowed. Also accept the rest of the JSON5 whitespace: vertical tab, form feed, U+FEFF, the Unicode space separators, and the line terminators U+2028 and U+2029, which also end a line comment. Non-ASCII whitespace is only recognized in UTF-8 input. Only the recursive-descent parser supports json5; the incremental parser and its constructor are unchanged, and FORCE_JSON_PSTACK builds keep using the recursive-descent parser for json5 input. With json5=false the lexer behaves exactly as before. Adds a test_json_parser_standalone test program, as the existing test binary only tests incremental parsing. --- src/backend/utils/adt/json.c | 2 +- src/backend/utils/adt/jsonb.c | 2 +- src/backend/utils/adt/jsonfuncs.c | 7 +- src/backend/utils/adt/pg_dependencies.c | 2 +- src/backend/utils/adt/pg_ndistinct.c | 2 +- src/common/jsonapi.c | 184 +++++++++- src/common/parse_manifest.c | 2 +- src/include/common/jsonapi.h | 5 +- src/interfaces/libpq-oauth/oauth-curl.c | 2 +- src/interfaces/libpq/fe-auth-oauth.c | 2 +- src/test/modules/test_escape/test_escape.c | 2 +- src/test/modules/test_json_parser/.gitignore | 2 + src/test/modules/test_json_parser/Makefile | 12 +- src/test/modules/test_json_parser/README | 8 +- .../test_json_parser/json5_comments.json5 | 12 + .../test_json_parser/json5_comments.out | 10 + src/test/modules/test_json_parser/meson.build | 36 +- .../test_json_parser/t/005_test_json5.pl | 153 ++++++++ .../test_json_parser/test_json_parser_perf.c | 2 +- .../test_json_parser_standalone.c | 337 ++++++++++++++++++ 20 files changed, 750 insertions(+), 34 deletions(-) create mode 100644 src/test/modules/test_json_parser/json5_comments.json5 create mode 100644 src/test/modules/test_json_parser/json5_comments.out create mode 100644 src/test/modules/test_json_parser/t/005_test_json5.pl create mode 100644 src/test/modules/test_json_parser/test_json_parser_standalone.c diff --git a/src/backend/utils/adt/json.c b/src/backend/utils/adt/json.c index 28e5f3cf9c0..7b48f3915c9 100644 --- a/src/backend/utils/adt/json.c +++ b/src/backend/utils/adt/json.c @@ -159,7 +159,7 @@ json_recv(PG_FUNCTION_ARGS) /* Validate it. */ makeJsonLexContextCstringLen(&lex, str, nbytes, GetDatabaseEncoding(), - false); + false, false); pg_parse_json_or_ereport(&lex, &nullSemAction); PG_RETURN_TEXT_P(cstring_to_text_with_len(str, nbytes)); diff --git a/src/backend/utils/adt/jsonb.c b/src/backend/utils/adt/jsonb.c index da0061ba1b8..12698e93437 100644 --- a/src/backend/utils/adt/jsonb.c +++ b/src/backend/utils/adt/jsonb.c @@ -245,7 +245,7 @@ jsonb_from_cstring(char *json, int len, bool unique_keys, Node *escontext) memset(&state, 0, sizeof(state)); memset(&sem, 0, sizeof(sem)); - makeJsonLexContextCstringLen(&lex, json, len, GetDatabaseEncoding(), true); + makeJsonLexContextCstringLen(&lex, json, len, GetDatabaseEncoding(), true, false); state.unique_keys = unique_keys; state.escontext = escontext; diff --git a/src/backend/utils/adt/jsonfuncs.c b/src/backend/utils/adt/jsonfuncs.c index bc7b556e22e..9ef170f6331 100644 --- a/src/backend/utils/adt/jsonfuncs.c +++ b/src/backend/utils/adt/jsonfuncs.c @@ -552,7 +552,8 @@ makeJsonLexContext(JsonLexContext *lex, text *json, bool need_escapes) VARDATA_ANY(json), VARSIZE_ANY_EXHDR(json), GetDatabaseEncoding(), - need_escapes); + need_escapes, + false); } /* @@ -2791,7 +2792,7 @@ populate_array_json(PopulateArrayContext *ctx, const char *json, int len) JsonSemAction sem; state.lex = makeJsonLexContextCstringLen(NULL, json, len, - GetDatabaseEncoding(), true); + GetDatabaseEncoding(), true, false); state.ctx = ctx; memset(&sem, 0, sizeof(sem)); @@ -3829,7 +3830,7 @@ get_json_object_as_hash(const char *json, int len, const char *funcname, state->function_name = funcname; state->hash = tab; state->lex = makeJsonLexContextCstringLen(NULL, json, len, - GetDatabaseEncoding(), true); + GetDatabaseEncoding(), true, false); sem->semstate = state; sem->array_start = hash_array_start; diff --git a/src/backend/utils/adt/pg_dependencies.c b/src/backend/utils/adt/pg_dependencies.c index 1b88c1dfc29..60f29e5b856 100644 --- a/src/backend/utils/adt/pg_dependencies.c +++ b/src/backend/utils/adt/pg_dependencies.c @@ -778,7 +778,7 @@ pg_dependencies_in(PG_FUNCTION_ARGS) sem_action.object_field_end = NULL; sem_action.scalar = dependencies_scalar; - lex = makeJsonLexContextCstringLen(NULL, str, strlen(str), PG_UTF8, true); + lex = makeJsonLexContextCstringLen(NULL, str, strlen(str), PG_UTF8, true, false); result = pg_parse_json(lex, &sem_action); freeJsonLexContext(lex); diff --git a/src/backend/utils/adt/pg_ndistinct.c b/src/backend/utils/adt/pg_ndistinct.c index 52a9eea6a59..280be715cc6 100644 --- a/src/backend/utils/adt/pg_ndistinct.c +++ b/src/backend/utils/adt/pg_ndistinct.c @@ -756,7 +756,7 @@ pg_ndistinct_in(PG_FUNCTION_ARGS) sem_action.scalar = ndistinct_scalar; lex = makeJsonLexContextCstringLen(NULL, str, strlen(str), - PG_UTF8, true); + PG_UTF8, true, false); result = pg_parse_json(lex, &sem_action); freeJsonLexContext(lex); diff --git a/src/common/jsonapi.c b/src/common/jsonapi.c index d3860197dad..81410d62bbf 100644 --- a/src/common/jsonapi.c +++ b/src/common/jsonapi.c @@ -18,6 +18,7 @@ #endif #include "common/jsonapi.h" +#include "common/unicode_category.h" #include "mb/pg_wchar.h" #include "port/pg_lfind.h" @@ -330,6 +331,140 @@ lex_expect(JsonParseContext ctx, JsonLexContext *lex, JsonTokenType token) (c) == '_' || \ IS_HIGHBIT_SET(c)) +/* + * Decode the UTF-8 character at s into *c, returning its length in bytes, or + * 0 if the input isn't UTF-8 or s doesn't start a valid UTF-8 character. + * json5 only gives meaning to non-ASCII characters outside of strings when + * it can decode them. + */ +static int +json5_decode_utf8(const JsonLexContext *lex, const char *s, const char *end, + char32_t *c) +{ + int len; + + if (lex->input_encoding != PG_UTF8) + return 0; + len = pg_utf_mblen((const unsigned char *) s); + if (len > end - s || !pg_utf8_islegal((const unsigned char *) s, len)) + return 0; + *c = utf8_to_unicode((const unsigned char *) s); + return len; +} + +/* + * Is the character at s a json5 line terminator other than LF or CR, that + * is U+2028 LINE SEPARATOR or U+2029 PARAGRAPH SEPARATOR? Returns its + * length in bytes, or 0. + */ +static inline int +json5_line_separator_len(const JsonLexContext *lex, const char *s, + const char *end) +{ + if (lex->input_encoding == PG_UTF8 && end - s >= 3 && + (unsigned char) s[0] == 0xE2 && (unsigned char) s[1] == 0x80 && + ((unsigned char) s[2] == 0xA8 || (unsigned char) s[2] == 0xA9)) + return 3; + return 0; +} + +/* + * Skip json5 whitespace and comments starting at s, returning the position + * after them, or NULL after setting up lex to report an unterminated block + * comment. Returning the position rather than taking a pointer to the + * caller's lexing cursor lets json_lex() keep that in a register. + * + * Besides the JSON whitespace characters, json5 allows vertical tab, form + * feed, U+FEFF, any Unicode space separator (Zs) and the line terminators + * U+2028 and U+2029. Comments count as whitespace, so this alternates + * between the two until neither matches. json5 is never used with + * incremental parsing, so a comment can't span chunks. + */ +static pg_noinline const char * +json5_skip_whitespace(JsonLexContext *lex, const char *s) +{ + const char *const end = lex->input + lex->input_length; + + while (s < end) + { + char32_t c; + int len; + + if (*s == ' ' || *s == '\t' || *s == '\r' || *s == '\v' || + *s == '\f') + s++; + else if (*s == '\n') + { + s++; + ++lex->line_number; + lex->line_start = s; + } + else if (IS_HIGHBIT_SET(*s)) + { + len = json5_decode_utf8(lex, s, end, &c); + if (len == 0) + break; + if (c != 0xFEFF) + { + pg_unicode_category cat = unicode_category(c); + + if (cat != PG_U_SPACE_SEPARATOR && + cat != PG_U_LINE_SEPARATOR && + cat != PG_U_PARAGRAPH_SEPARATOR) + break; + } + s += len; + } + else if (*s == '/' && s + 1 < end && *(s + 1) == '/') + { + /* + * Line comment. The terminator, if any, is consumed as + * whitespace on the next round; at end of input the comment is + * implicitly closed. + */ + s += 2; + while (s < end && *s != '\n' && *s != '\r' && + json5_line_separator_len(lex, s, end) == 0) + s++; + } + else if (*s == '/' && s + 1 < end && *(s + 1) == '*') + { + /* block comment; star tracks a possible closing '*' */ + bool star = false; + bool closed = false; + + s += 2; + while (s < end) + { + char ch = *s++; + + if (star && ch == '/') + { + closed = true; + break; + } + star = (ch == '*'); + if (ch == '\n') + { + ++lex->line_number; + lex->line_start = s; + } + } + if (!closed) + { + lex->token_start = s; + lex->prev_token_terminator = lex->token_terminator; + lex->token_terminator = s; + return NULL; + } + } + else + break; /* includes a lone '/', an invalid token */ + } + + return s; +} + /* * Utility function to check if a string is a valid JSON number. * @@ -390,7 +525,8 @@ IsValidJsonNumber(const char *str, size_t len) */ JsonLexContext * makeJsonLexContextCstringLen(JsonLexContext *lex, const char *json, - size_t len, int encoding, bool need_escapes) + size_t len, int encoding, bool need_escapes, + bool json5) { if (lex == NULL) { @@ -408,6 +544,7 @@ makeJsonLexContextCstringLen(JsonLexContext *lex, const char *json, lex->input_length = len; lex->input_encoding = encoding; lex->need_escapes = need_escapes; + lex->json5 = json5; if (need_escapes) { /* @@ -739,27 +876,32 @@ freeJsonLexContext(JsonLexContext *lex) * JSON parser. This is a useful way to validate that it's doing the right * thing at least for non-incremental cases. If this is on we expect to see * regression diffs relating to error messages about stack depth, but no - * other differences. + * other differences. json5 input is exempt, as only the recursive descent + * parser knows that syntax. */ JsonParseErrorType pg_parse_json(JsonLexContext *lex, const JsonSemAction *sem) { -#ifdef FORCE_JSON_PSTACK - /* - * We don't need partial token processing, there is only one chunk. But we - * still need to init the partial token string so that freeJsonLexContext - * works, so perform the full incremental initialization. - */ - if (!allocate_incremental_state(lex)) - return JSON_OUT_OF_MEMORY; - - return pg_parse_json_incremental(lex, sem, lex->input, lex->input_length, true); - -#else - JsonTokenType tok; JsonParseErrorType result; +#ifdef FORCE_JSON_PSTACK + if (!lex->json5) + { + /* + * We don't need partial token processing, there is only one chunk. + * But we still need to init the partial token string so that + * freeJsonLexContext works, so perform the full incremental + * initialization. + */ + if (!allocate_incremental_state(lex)) + return JSON_OUT_OF_MEMORY; + + return pg_parse_json_incremental(lex, sem, lex->input, + lex->input_length, true); + } +#endif + if (lex == &failed_oom) return JSON_OUT_OF_MEMORY; if (lex->incremental) @@ -789,7 +931,6 @@ pg_parse_json(JsonLexContext *lex, const JsonSemAction *sem) result = lex_expect(JSON_PARSE_END, lex, JSON_TOKEN_END); return result; -#endif } /* @@ -1858,6 +1999,15 @@ json_lex(JsonLexContext *lex) lex->line_start = s; } } + + /* json5 also allows more whitespace characters, and comments */ + if (unlikely(lex->json5)) + { + s = json5_skip_whitespace(lex, s); + if (s == NULL) + return JSON_UNTERMINATED_COMMENT; + } + lex->token_start = s; /* Determine token type. */ @@ -2551,6 +2701,8 @@ json_errdetail(JsonParseErrorType error, JsonLexContext *lex) return _("Unicode high surrogate must not follow a high surrogate."); case JSON_UNICODE_LOW_SURROGATE: return _("Unicode low surrogate must follow a high surrogate."); + case JSON_UNTERMINATED_COMMENT: + return _("Block comment is not terminated."); case JSON_SEM_ACTION_FAILED: /* fall through to the error code after switch */ break; diff --git a/src/common/parse_manifest.c b/src/common/parse_manifest.c index 565bdd01bee..a1762c6775e 100644 --- a/src/common/parse_manifest.c +++ b/src/common/parse_manifest.c @@ -238,7 +238,7 @@ json_parse_manifest(JsonManifestParseContext *context, const char *buffer, parse.saw_version_field = false; /* Create a JSON lexing context. */ - lex = makeJsonLexContextCstringLen(NULL, buffer, size, PG_UTF8, true); + lex = makeJsonLexContextCstringLen(NULL, buffer, size, PG_UTF8, true, false); /* Set up semantic actions. */ sem.semstate = &parse; diff --git a/src/include/common/jsonapi.h b/src/include/common/jsonapi.h index 85cc9a11d97..d7cc2ac43e1 100644 --- a/src/include/common/jsonapi.h +++ b/src/include/common/jsonapi.h @@ -56,6 +56,7 @@ typedef enum JsonParseErrorType JSON_UNICODE_UNTRANSLATABLE, JSON_UNICODE_HIGH_SURROGATE, JSON_UNICODE_LOW_SURROGATE, + JSON_UNTERMINATED_COMMENT, JSON_SEM_ACTION_FAILED, /* error should already be reported */ } JsonParseErrorType; @@ -106,6 +107,7 @@ typedef struct JsonLexContext const char *token_terminator; const char *prev_token_terminator; bool incremental; + bool json5; JsonTokenType token_type; int lex_level; uint32 flags; @@ -219,7 +221,8 @@ extern JsonLexContext *makeJsonLexContextCstringLen(JsonLexContext *lex, const char *json, size_t len, int encoding, - bool need_escapes); + bool need_escapes, + bool json5); /* * make a JsonLexContext suitable for incremental parsing. diff --git a/src/interfaces/libpq-oauth/oauth-curl.c b/src/interfaces/libpq-oauth/oauth-curl.c index 9e0d39773e5..7b94db7dfac 100644 --- a/src/interfaces/libpq-oauth/oauth-curl.c +++ b/src/interfaces/libpq-oauth/oauth-curl.c @@ -897,7 +897,7 @@ parse_oauth_json(struct async_ctx *actx, const struct json_field *fields) return false; } - makeJsonLexContextCstringLen(&lex, resp->data, resp->len, PG_UTF8, true); + makeJsonLexContextCstringLen(&lex, resp->data, resp->len, PG_UTF8, true, false); setJsonLexContextOwnsTokens(&lex, true); /* must not leak on error */ ctx.errbuf = &actx->errbuf; diff --git a/src/interfaces/libpq/fe-auth-oauth.c b/src/interfaces/libpq/fe-auth-oauth.c index 826f7461cb3..c8cc0f0cf70 100644 --- a/src/interfaces/libpq/fe-auth-oauth.c +++ b/src/interfaces/libpq/fe-auth-oauth.c @@ -547,7 +547,7 @@ handle_oauth_sasl_error(PGconn *conn, const char *msg, int msglen) return false; } - lex = makeJsonLexContextCstringLen(NULL, msg, msglen, PG_UTF8, true); + lex = makeJsonLexContextCstringLen(NULL, msg, msglen, PG_UTF8, true, false); setJsonLexContextOwnsTokens(lex, true); /* must not leak on error */ initPQExpBuffer(&ctx.errbuf); diff --git a/src/test/modules/test_escape/test_escape.c b/src/test/modules/test_escape/test_escape.c index 4b556f73891..4fcfd7f4b92 100644 --- a/src/test/modules/test_escape/test_escape.c +++ b/src/test/modules/test_escape/test_escape.c @@ -235,7 +235,7 @@ test_gb18030_json(pe_test_config *tc) /* test itself */ lex = makeJsonLexContextCstringLen(NULL, raw_buf->data, input_len, - PG_GB18030, false); + PG_GB18030, false, false); json_error = pg_parse_json(lex, &sem); report_result(tc, json_error == JSON_UNICODE_ESCAPE_FORMAT, testname->data, "", diff --git a/src/test/modules/test_json_parser/.gitignore b/src/test/modules/test_json_parser/.gitignore index f032d1e4f90..6c0ac0fd557 100644 --- a/src/test/modules/test_json_parser/.gitignore +++ b/src/test/modules/test_json_parser/.gitignore @@ -2,3 +2,5 @@ tmp_check test_json_parser_perf test_json_parser_incremental test_json_parser_incremental_shlib +test_json_parser_standalone +test_json_parser_standalone_shlib diff --git a/src/test/modules/test_json_parser/Makefile b/src/test/modules/test_json_parser/Makefile index af3f19424ed..2a10778dd57 100644 --- a/src/test/modules/test_json_parser/Makefile +++ b/src/test/modules/test_json_parser/Makefile @@ -4,9 +4,9 @@ PGAPPICON = win32 TAP_TESTS = 1 -OBJS = test_json_parser_incremental.o test_json_parser_perf.o $(WIN32RES) +OBJS = test_json_parser_incremental.o test_json_parser_perf.o test_json_parser_standalone.o $(WIN32RES) -EXTRA_CLEAN = test_json_parser_incremental$(X) test_json_parser_incremental_shlib$(X) test_json_parser_perf$(X) +EXTRA_CLEAN = test_json_parser_incremental$(X) test_json_parser_incremental_shlib$(X) test_json_parser_perf$(X) test_json_parser_standalone$(X) test_json_parser_standalone_shlib$(X) ifdef USE_PGXS PG_CONFIG = pg_config @@ -19,7 +19,7 @@ include $(top_builddir)/src/Makefile.global include $(top_srcdir)/contrib/contrib-global.mk endif -all: test_json_parser_incremental$(X) test_json_parser_incremental_shlib$(X) test_json_parser_perf$(X) +all: test_json_parser_incremental$(X) test_json_parser_incremental_shlib$(X) test_json_parser_perf$(X) test_json_parser_standalone$(X) test_json_parser_standalone_shlib$(X) %.o: $(top_srcdir)/$(subdir)/%.c @@ -32,6 +32,12 @@ test_json_parser_incremental_shlib$(X): test_json_parser_incremental.o $(WIN32RE test_json_parser_perf$(X): test_json_parser_perf.o $(WIN32RES) $(CC) $(CFLAGS) $^ $(PG_LIBS_INTERNAL) $(LDFLAGS) $(LDFLAGS_EX) $(PG_LIBS) $(LIBS) -o $@ +test_json_parser_standalone$(X): test_json_parser_standalone.o $(WIN32RES) + $(CC) $(CFLAGS) $^ $(PG_LIBS_INTERNAL) $(LDFLAGS) $(LDFLAGS_EX) $(PG_LIBS) $(LIBS) -o $@ + +test_json_parser_standalone_shlib$(X): test_json_parser_standalone.o $(WIN32RES) + $(CC) $(CFLAGS) $^ $(LDFLAGS) -lpgcommon_excluded_shlib $(libpq_pgport_shlib) $(filter -lintl, $(LIBS)) -o $@ + speed-check: test_json_parser_perf$(X) @echo Standard parser: time ./test_json_parser_perf 10000 $(top_srcdir)/$(subdir)/tiny.json diff --git a/src/test/modules/test_json_parser/README b/src/test/modules/test_json_parser/README index 61e7c78d588..e56e8273f94 100644 --- a/src/test/modules/test_json_parser/README +++ b/src/test/modules/test_json_parser/README @@ -1,7 +1,7 @@ Module `test_json_parser` ========================= -This module contains two programs for testing the json parsers. +This module contains three programs for testing the json parsers. - `test_json_parser_incremental` is for testing the incremental parser, It reads in a file and passes it in very small chunks (default is 60 bytes at a @@ -12,6 +12,12 @@ This module contains two programs for testing the json parsers. using semantic routines. The semantic routines re-output the json, although not in a very pretty form. The required non-option argument is the input file name. +- `test_json_parser_standalone` is for testing the standalone (recursive + descent) parser. It reads the whole input file and parses it in a single + call. The option "--json5" makes the parser accept JSON5 input; only this + parser supports JSON5. The "-s" and "-o" options work as for + `test_json_parser_incremental`. The required non-option argument is the + input file name. - `test_json_parser_perf` is for speed testing both the standard recursive descent parser and the non-recursive incremental parser. If given the `-i` flag it uses the non-recursive parser, diff --git a/src/test/modules/test_json_parser/json5_comments.json5 b/src/test/modules/test_json_parser/json5_comments.json5 new file mode 100644 index 00000000000..9cc9e539f94 --- /dev/null +++ b/src/test/modules/test_json_parser/json5_comments.json5 @@ -0,0 +1,12 @@ +// leading line comment +{ + /* block comment before key */ "key1": "value1", // trailing line comment on a pair + "key2": /* inline block comment */ "value2", + "arr": [ + 1, + /* mid-array block comment */ + 2, + 3 + ] +} +/* trailing block comment */ diff --git a/src/test/modules/test_json_parser/json5_comments.out b/src/test/modules/test_json_parser/json5_comments.out new file mode 100644 index 00000000000..1ad7126976e --- /dev/null +++ b/src/test/modules/test_json_parser/json5_comments.out @@ -0,0 +1,10 @@ +{ +"key1": "value1", +"key2": "value2", +"arr": [ +1, +2, +3 +] + +} diff --git a/src/test/modules/test_json_parser/meson.build b/src/test/modules/test_json_parser/meson.build index 2688686e37b..80692465be7 100644 --- a/src/test/modules/test_json_parser/meson.build +++ b/src/test/modules/test_json_parser/meson.build @@ -31,6 +31,37 @@ test_json_parser_incremental_shlib = executable('test_json_parser_incremental_sh }, ) +test_json_parser_standalone_sources = files( + 'test_json_parser_standalone.c', +) + +if host_system == 'windows' + test_json_parser_standalone_sources += rc_bin_gen.process(win32ver_rc, extra_args: [ + '--NAME', 'test_json_parser_standalone', + '--FILEDESC', 'standalone json parser tester', + ]) +endif + +test_json_parser_standalone = executable('test_json_parser_standalone', + test_json_parser_standalone_sources, + dependencies: [frontend_code], + kwargs: default_bin_args + { + 'install': false, + }, +) + +# A second version of test_json_parser_standalone, this time compiled against +# the shared-library flavor of jsonapi. +test_json_parser_standalone_shlib = executable('test_json_parser_standalone_shlib', + test_json_parser_standalone_sources, + dependencies: [frontend_shlib_code, libpq], + c_args: ['-DJSONAPI_SHLIB_ALLOC'], + link_with: [common_excluded_shlib], + kwargs: default_bin_args + { + 'install': false, + }, +) + test_json_parser_perf_sources = files( 'test_json_parser_perf.c', ) @@ -59,12 +90,15 @@ tests += { 't/001_test_json_parser_incremental.pl', 't/002_inline.pl', 't/003_test_semantic.pl', - 't/004_test_parser_perf.pl' + 't/004_test_parser_perf.pl', + 't/005_test_json5.pl' ], 'deps': [ test_json_parser_incremental, test_json_parser_incremental_shlib, test_json_parser_perf, + test_json_parser_standalone, + test_json_parser_standalone_shlib, ], }, } diff --git a/src/test/modules/test_json_parser/t/005_test_json5.pl b/src/test/modules/test_json_parser/t/005_test_json5.pl new file mode 100644 index 00000000000..815d4699306 --- /dev/null +++ b/src/test/modules/test_json_parser/t/005_test_json5.pl @@ -0,0 +1,153 @@ + +# Copyright (c) 2021-2026, PostgreSQL Global Development Group + +# Test JSON5 support in the standalone (recursive descent) JSON parser. +# Each feature fixture must be accepted in --json5 mode with the +# expected semantic output, and rejected without --json5. The +# incremental parser does not support JSON5. + +use strict; +use warnings FATAL => 'all'; + +use PostgreSQL::Test::Utils; +use Test::More; +use FindBin; +use File::Temp qw(tempfile); + +my $dir = PostgreSQL::Test::Utils::tempdir; + +my @exes = ( + [ "test_json_parser_standalone", ], + [ "test_json_parser_standalone", "-o", ], + [ "test_json_parser_standalone_shlib", ], + [ "test_json_parser_standalone_shlib", "-o", ]); + +# Inline json5 inputs with their expected semantic output, or undef for +# inputs that must be rejected in json5 mode. Non-ASCII input is given +# as UTF-8 bytes. +my @json5_inline_cases = ( + [ 'leading no-break space', "\xc2\xa0[1]", "[\n1\n]" ], + [ 'leading byte order mark', "\xef\xbb\xbf[1]", "[\n1\n]" ], + [ 'vertical tab and form feed', "[1,\x0b2,\f3]", "[\n1,\n2,\n3\n]" ], + [ 'ideographic space', "[1,\xe3\x80\x80 2]", "[\n1,\n2\n]" ], + [ 'line comment ended by end of input', "1 // c", "1" ], + [ + 'comment markers inside a string', + "[\"a /* b */ c // d\"]", "[\n\"a /* b */ c // d\"\n]" + ], + [ 'block comment only', "/* c */", undef ], + [ 'line comment only', "// c", undef ], + [ + 'line comment ended by U+2028', + "[1, // c\xe2\x80\xa8 2]", "[\n1,\n2\n]" + ], + [ + 'line comment ended by U+2029', + "[1, // c\xe2\x80\xa9 2]", "[\n1,\n2\n]" + ], + [ 'non-space symbol outside strings', "[1,\xe2\x80\xa6 2]", undef ], + [ 'invalid UTF-8 outside strings', "[1,\xc2 2]", undef ],); + +# One entry per feature: fixture basename plus, where the failing token +# is predictable, a regex for the error reported without --json5. +my @features = ( + { + name => 'comments', + file => 'json5_comments', + error => qr/Token "\/" is invalid/, + },); + +# Run the parser executable @$exe with @args and return its output, with +# the CRLF line ends that printf() produces on Windows normalized to LF. +sub run_parser +{ + my ($exe, @args) = @_; + my ($stdout, $stderr) = run_command([ @$exe, @args ]); + + $stdout =~ s/\r\n/\n/g if $windows_os; + return ($stdout, $stderr); +} + +# Parse $file with --json5 and compare the semantic output against +# $expected. +sub check_accepted +{ + my ($exe, $file, $expected, $label) = @_; + + my ($stdout, $stderr) = run_parser($exe, "-s", "--json5", $file); + + is($stderr, "", "$label: no error output"); + + my ($fh, $fname) = tempfile(DIR => $dir); + print $fh $stdout, "\n"; + close($fh); + + my @diffopts = ("-u"); + push(@diffopts, "--strip-trailing-cr") if $windows_os; + ($stdout, $stderr) = + run_command([ "diff", @diffopts, $fname, $expected ]); + + is($stdout, "", "$label: no output diff"); + is($stderr, "", "$label: no diff error"); +} + +# Check that parsing $file fails, matching $error on stderr if given. +sub check_rejected +{ + my ($exe, $file, $label, $error, @flags) = @_; + + my ($stdout, $stderr) = run_parser($exe, "-s", @flags, $file); + + unlike($stdout, qr/SUCCESS/, "$label: parsing fails"); + if (defined $error) + { + like($stderr, $error, "$label: correct error output"); + } + else + { + isnt($stderr, "", "$label: error output"); + } +} + +foreach my $exe (@exes) +{ + note "testing executable @$exe"; + + foreach my $c (@json5_inline_cases) + { + my ($label, $content, $expected) = @$c; + my ($fh, $fname) = tempfile(DIR => $dir); + + # binary mode, so that CR and LF in $content reach the parser as + # they are + binmode($fh); + print $fh $content; + close($fh); + + if (defined $expected) + { + my ($stdout, $stderr) = + run_parser($exe, "-s", "--json5", $fname); + + is($stdout, $expected, "json5 mode: $label: output"); + is($stderr, "", "json5 mode: $label: no error output"); + } + else + { + check_rejected($exe, $fname, "json5 mode: $label", undef, + "--json5"); + } + } + + foreach my $f (@features) + { + my $file = "$FindBin::RealBin/../$f->{file}.json5"; + my $expected = "$FindBin::RealBin/../$f->{file}.out"; + + check_accepted($exe, $file, $expected, "json5 $f->{name}"); + check_rejected($exe, $file, "non-json5 mode: $f->{name}", + $f->{error}); + } +} + +done_testing(); diff --git a/src/test/modules/test_json_parser/test_json_parser_perf.c b/src/test/modules/test_json_parser/test_json_parser_perf.c index 9786263a191..d92e8e875ec 100644 --- a/src/test/modules/test_json_parser/test_json_parser_perf.c +++ b/src/test/modules/test_json_parser/test_json_parser_perf.c @@ -76,7 +76,7 @@ main(int argc, char **argv) else { lex = makeJsonLexContextCstringLen(NULL, json.data, json.len, - PG_UTF8, false); + PG_UTF8, false, false); result = pg_parse_json(lex, &nullSemAction); freeJsonLexContext(lex); } diff --git a/src/test/modules/test_json_parser/test_json_parser_standalone.c b/src/test/modules/test_json_parser/test_json_parser_standalone.c new file mode 100644 index 00000000000..451d48a4d40 --- /dev/null +++ b/src/test/modules/test_json_parser/test_json_parser_standalone.c @@ -0,0 +1,337 @@ +/*------------------------------------------------------------------------- + * + * test_json_parser_standalone.c + * Test program for the standalone (recursive descent) JSON parser + * + * Copyright (c) 2024-2026, PostgreSQL Global Development Group + * + * IDENTIFICATION + * src/test/modules/test_json_parser/test_json_parser_standalone.c + * + * This program tests the standalone (non-incremental) JSON parser. The + * whole input file is read into memory and parsed in a single call. This + * is the test harness for JSON5 mode, which only this parser supports. + * + * If the -s flag is given, the program does semantic processing. This should + * just mirror back the json, albeit with white space changes. + * + * If the -o flag is given, the JSONLEX_CTX_OWNS_TOKENS flag is set. (This can + * be used in combination with a leak sanitizer; without the option, the parser + * may leak memory with invalid JSON.) + * + * If the --json5 flag is given, the input is parsed as JSON5. + * + * The argument specifies the file containing the JSON input. + * + *------------------------------------------------------------------------- + */ + +#include "postgres_fe.h" + +#include +#include +#include +#include + +#include "common/jsonapi.h" +#include "common/logging.h" +#include "getopt_long.h" +#include "lib/stringinfo.h" +#include "mb/pg_wchar.h" + +typedef struct DoState +{ + JsonLexContext *lex; + bool elem_is_first; + StringInfo buf; +} DoState; + +static void usage(const char *progname); +static void escape_json(StringInfo buf, const char *str); + +/* semantic action functions for parser */ +static JsonParseErrorType do_object_start(void *state); +static JsonParseErrorType do_object_end(void *state); +static JsonParseErrorType do_object_field_start(void *state, char *fname, bool isnull); +static JsonParseErrorType do_object_field_end(void *state, char *fname, bool isnull); +static JsonParseErrorType do_array_start(void *state); +static JsonParseErrorType do_array_end(void *state); +static JsonParseErrorType do_array_element_start(void *state, bool isnull); +static JsonParseErrorType do_array_element_end(void *state, bool isnull); +static JsonParseErrorType do_scalar(void *state, char *token, JsonTokenType tokentype); + +static JsonSemAction sem = { + .object_start = do_object_start, + .object_end = do_object_end, + .object_field_start = do_object_field_start, + .object_field_end = do_object_field_end, + .array_start = do_array_start, + .array_end = do_array_end, + .array_element_start = do_array_element_start, + .array_element_end = do_array_element_end, + .scalar = do_scalar +}; + +static bool lex_owns_tokens = false; + +int +main(int argc, char **argv) +{ + FILE *json_file; + JsonParseErrorType result; + JsonLexContext *lex; + StringInfoData json; + size_t n_read; + struct stat statbuf; + const JsonSemAction *testsem = &nullSemAction; + char *testfile; + int c; + bool need_strings = false; + bool json5 = false; + int ret = 0; + + static const struct option long_options[] = { + {"json5", no_argument, NULL, '5'}, + {NULL, 0, NULL, 0}, + }; + + pg_logging_init(argv[0]); + + lex = calloc(1, sizeof(JsonLexContext)); + if (!lex) + pg_fatal("out of memory"); + + while ((c = getopt_long(argc, argv, "os", long_options, NULL)) != -1) + { + switch (c) + { + case 'o': /* switch token ownership */ + lex_owns_tokens = true; + break; + case 's': /* do semantic processing */ + testsem = &sem; + sem.semstate = palloc_object(struct DoState); + ((struct DoState *) sem.semstate)->lex = lex; + ((struct DoState *) sem.semstate)->buf = makeStringInfo(); + need_strings = true; + break; + case '5': /* parse as json5 */ + json5 = true; + break; + } + } + + if (optind < argc) + { + testfile = argv[optind]; + optind++; + } + else + { + usage(argv[0]); + exit(1); + } + + if ((json_file = fopen(testfile, PG_BINARY_R)) == NULL) + pg_fatal("error opening input: %m"); + + if (fstat(fileno(json_file), &statbuf) != 0) + pg_fatal("error statting input: %m"); + + initStringInfo(&json); + enlargeStringInfo(&json, statbuf.st_size); + n_read = fread(json.data, 1, statbuf.st_size, json_file); + if (n_read < (size_t) statbuf.st_size) + pg_fatal("error reading input file: %d", ferror(json_file)); + json.len = n_read; + json.data[json.len] = '\0'; + fclose(json_file); + + makeJsonLexContextCstringLen(lex, json.data, json.len, PG_UTF8, + need_strings, json5); + setJsonLexContextOwnsTokens(lex, lex_owns_tokens); + + result = pg_parse_json(lex, testsem); + if (result != JSON_SUCCESS) + { + fprintf(stderr, "%s\n", json_errdetail(result, lex)); + ret = 1; + } + else if (!need_strings) + printf("SUCCESS!\n"); + + freeJsonLexContext(lex); + free(json.data); + free(lex); + + return ret; +} + +/* + * The semantic routines here essentially just output the same json, except + * for white space. We could pretty print it but there's no need for our + * purposes. The result should be able to be fed to any JSON processor + * such as jq for validation. + */ + +static JsonParseErrorType +do_object_start(void *state) +{ + DoState *_state = (DoState *) state; + + printf("{\n"); + _state->elem_is_first = true; + + return JSON_SUCCESS; +} + +static JsonParseErrorType +do_object_end(void *state) +{ + DoState *_state = (DoState *) state; + + printf("\n}\n"); + _state->elem_is_first = false; + + return JSON_SUCCESS; +} + +static JsonParseErrorType +do_object_field_start(void *state, char *fname, bool isnull) +{ + DoState *_state = (DoState *) state; + + if (!_state->elem_is_first) + printf(",\n"); + resetStringInfo(_state->buf); + escape_json(_state->buf, fname); + printf("%s: ", _state->buf->data); + _state->elem_is_first = false; + + return JSON_SUCCESS; +} + +static JsonParseErrorType +do_object_field_end(void *state, char *fname, bool isnull) +{ + if (!lex_owns_tokens) + free(fname); + + return JSON_SUCCESS; +} + +static JsonParseErrorType +do_array_start(void *state) +{ + DoState *_state = (DoState *) state; + + printf("[\n"); + _state->elem_is_first = true; + + return JSON_SUCCESS; +} + +static JsonParseErrorType +do_array_end(void *state) +{ + DoState *_state = (DoState *) state; + + printf("\n]\n"); + _state->elem_is_first = false; + + return JSON_SUCCESS; +} + +static JsonParseErrorType +do_array_element_start(void *state, bool isnull) +{ + DoState *_state = (DoState *) state; + + if (!_state->elem_is_first) + printf(",\n"); + _state->elem_is_first = false; + + return JSON_SUCCESS; +} + +static JsonParseErrorType +do_array_element_end(void *state, bool isnull) +{ + /* nothing to do */ + + return JSON_SUCCESS; +} + +static JsonParseErrorType +do_scalar(void *state, char *token, JsonTokenType tokentype) +{ + DoState *_state = (DoState *) state; + + if (tokentype == JSON_TOKEN_STRING) + { + resetStringInfo(_state->buf); + escape_json(_state->buf, token); + printf("%s", _state->buf->data); + } + else + printf("%s", token); + + if (!lex_owns_tokens) + free(token); + + return JSON_SUCCESS; +} + + +/* copied from backend code */ +static void +escape_json(StringInfo buf, const char *str) +{ + const char *p; + + appendStringInfoCharMacro(buf, '"'); + for (p = str; *p; p++) + { + switch (*p) + { + case '\b': + appendStringInfoString(buf, "\\b"); + break; + case '\f': + appendStringInfoString(buf, "\\f"); + break; + case '\n': + appendStringInfoString(buf, "\\n"); + break; + case '\r': + appendStringInfoString(buf, "\\r"); + break; + case '\t': + appendStringInfoString(buf, "\\t"); + break; + case '"': + appendStringInfoString(buf, "\\\""); + break; + case '\\': + appendStringInfoString(buf, "\\\\"); + break; + default: + if ((unsigned char) *p < ' ') + appendStringInfo(buf, "\\u%04x", (int) *p); + else + appendStringInfoCharMacro(buf, *p); + break; + } + } + appendStringInfoCharMacro(buf, '"'); +} + +static void +usage(const char *progname) +{ + fprintf(stderr, "Usage: %s [OPTION ...] testfile\n", progname); + fprintf(stderr, "Options:\n"); + fprintf(stderr, " -o set JSONLEX_CTX_OWNS_TOKENS for leak checking\n"); + fprintf(stderr, " -s do semantic processing\n"); + fprintf(stderr, " --json5 parse input as json5\n"); +} -- 2.55.0