diff --git a/docs/mkdocs/docs/api/basic_json/dump.md b/docs/mkdocs/docs/api/basic_json/dump.md index a3c9db3a8..18b516a9e 100644 --- a/docs/mkdocs/docs/api/basic_json/dump.md +++ b/docs/mkdocs/docs/api/basic_json/dump.md @@ -25,10 +25,15 @@ and `ensure_ascii` parameters. result consists of ASCII characters only. `error_handler` (in) -: how to react on decoding errors; there are three possible values (see [`error_handler_t`](error_handler_t.md): - `strict` (throws an exception in case a decoding error occurs; default), `replace` (replace invalid UTF-8 sequences - with U+FFFD), and `ignore` (ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the - output unchanged, and invalid bytes are dropped)). +: how to react on decoding errors; there are four possible values (see [`error_handler_t`](error_handler_t.md)): + + - `strict`: throw a [`type_error`](../../home/exceptions.md#type-errors) exception in case a decoding error occurs + (default), + - `replace`: replace invalid UTF-8 sequences with U+FFFD (� REPLACEMENT CHARACTER), + - `ignore`: ignore invalid UTF-8 sequences during serialization; all valid bytes are copied to the output unchanged, + and invalid bytes are dropped, and + - `keep`: keep invalid UTF-8 sequences during serialization; all bytes are copied to the output unchanged, so the + result is not valid UTF-8. ## Return value @@ -94,3 +99,4 @@ Binary values are serialized as an object containing two keys: - Indentation character `indent_char`, option `ensure_ascii` and exceptions added in version 3.0.0. - Error handlers added in version 3.4.0. - Serialization of binary values added in version 3.8.0. +- Error handler value `keep` added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json/error_handler_t.md b/docs/mkdocs/docs/api/basic_json/error_handler_t.md index f20c33c03..5fedfe9f8 100644 --- a/docs/mkdocs/docs/api/basic_json/error_handler_t.md +++ b/docs/mkdocs/docs/api/basic_json/error_handler_t.md @@ -4,12 +4,13 @@ enum class error_handler_t { strict, replace, - ignore + ignore, + keep }; ``` This enumeration is used in the [`dump`](dump.md) function to choose how to treat decoding errors while serializing a -`basic_json` value. Three values are differentiated: +`basic_json` value. Four values are differentiated: strict : throw a `type_error` exception in case of invalid UTF-8 @@ -20,6 +21,12 @@ replace ignore : ignore invalid UTF-8 sequences; all valid bytes are copied to the output unchanged, and invalid bytes are dropped +keep +: keep invalid UTF-8 sequences; all bytes are copied to the output unchanged. Valid characters are still escaped as + usual (e.g., `"`, `\\`, and control characters), so the result has valid JSON syntax, but it is not valid UTF-8. + In particular, [`parse`](parse.md) rejects it, and with `ensure_ascii` set to `true`, the invalid bytes are the + only non-ASCII bytes of the output. + ## Examples ??? example @@ -40,3 +47,4 @@ ignore ## Version history - Added in version 3.4.0. +- Added value `keep` in version 3.13.0. diff --git a/docs/mkdocs/docs/examples/error_handler_t.cpp b/docs/mkdocs/docs/examples/error_handler_t.cpp index b4718d7e6..e5090aaec 100644 --- a/docs/mkdocs/docs/examples/error_handler_t.cpp +++ b/docs/mkdocs/docs/examples/error_handler_t.cpp @@ -1,3 +1,4 @@ +#include #include #include @@ -21,4 +22,12 @@ int main() << "\nstring with ignored invalid characters: " << j_invalid.dump(-1, ' ', false, json::error_handler_t::ignore) << '\n'; + + // the invalid byte is kept; print the result byte-wise to make it visible + std::cout << "string with kept invalid characters:"; + for (const unsigned char c : j_invalid.dump(-1, ' ', false, json::error_handler_t::keep)) + { + std::cout << ' ' << std::hex << std::setw(2) << std::setfill('0') << static_cast(c); + } + std::cout << '\n'; } diff --git a/docs/mkdocs/docs/examples/error_handler_t.output b/docs/mkdocs/docs/examples/error_handler_t.output index 718d62bee..12c374dba 100644 --- a/docs/mkdocs/docs/examples/error_handler_t.output +++ b/docs/mkdocs/docs/examples/error_handler_t.output @@ -1,3 +1,4 @@ [json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9 string with replaced invalid characters: "ä�ü" string with ignored invalid characters: "äü" +string with kept invalid characters: 22 c3 a4 a9 c3 bc 22 diff --git a/docs/mkdocs/docs/features/serialization.md b/docs/mkdocs/docs/features/serialization.md index a875fcdde..b74e10cf4 100644 --- a/docs/mkdocs/docs/features/serialization.md +++ b/docs/mkdocs/docs/features/serialization.md @@ -64,6 +64,7 @@ serialization fails by default. The fourth argument of `dump` selects an - `strict` (default) — throw a [`type_error.316`](../home/exceptions.md#jsonexceptiontype_error316) exception. - `replace` — replace invalid bytes with the Unicode replacement character U+FFFD (`�`). - `ignore` — silently drop invalid bytes. +- `keep` — copy invalid bytes to the output unchanged; the result is not valid UTF-8. ??? example diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index ee76596f6..86da22b13 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -755,6 +755,7 @@ The `dump()` function only works with UTF-8 encoded strings; that is, if you ass - Pass an error handler as last parameter to the `dump()` function to avoid this exception: - `json::error_handler_t::replace` will replace invalid bytes sequences with `U+FFFD` - `json::error_handler_t::ignore` will silently ignore invalid byte sequences + - `json::error_handler_t::keep` will copy invalid byte sequences to the output unchanged ### json.exception.type_error.317 diff --git a/docs/mkdocs/docs/home/faq.md b/docs/mkdocs/docs/home/faq.md index 8b3602bd1..7340af63c 100644 --- a/docs/mkdocs/docs/home/faq.md +++ b/docs/mkdocs/docs/home/faq.md @@ -85,7 +85,7 @@ The library supports **Unicode input** as follows: - The library will not replace [Unicode noncharacters](http://www.unicode.org/faq/private_use.html#nonchar1). - Invalid surrogates (e.g., incomplete pairs such as `\uDEAD`) will yield parse errors. - The strings stored in the library are UTF-8 encoded. When using the default string type (`std::string`), note that its length/size functions return the number of stored bytes rather than the number of characters or glyphs. -- When you store strings with different encodings in the library, calling [`dump()`](https://nlohmann.github.io/json/classnlohmann_1_1basic__json_a50ec80b02d0f3f51130d4abb5d1cfdc5.html#a50ec80b02d0f3f51130d4abb5d1cfdc5) may throw an exception unless `json::error_handler_t::replace` or `json::error_handler_t::ignore` are used as error handlers. +- When you store strings with different encodings in the library, calling [`dump()`](https://nlohmann.github.io/json/classnlohmann_1_1basic__json_a50ec80b02d0f3f51130d4abb5d1cfdc5.html#a50ec80b02d0f3f51130d4abb5d1cfdc5) may throw an exception unless `json::error_handler_t::replace`, `json::error_handler_t::ignore`, or `json::error_handler_t::keep` are used as error handlers. In most cases, the parser is right to complain, because the input is not UTF-8 encoded. This is especially true for Microsoft Windows, where Latin-1 or ISO 8859-1 is often the standard encoding. diff --git a/include/nlohmann/detail/output/serializer.hpp b/include/nlohmann/detail/output/serializer.hpp index f968b6001..8eb038835 100644 --- a/include/nlohmann/detail/output/serializer.hpp +++ b/include/nlohmann/detail/output/serializer.hpp @@ -48,7 +48,8 @@ enum class error_handler_t { strict, ///< throw a type_error exception in case of invalid UTF-8 replace, ///< replace invalid UTF-8 sequences with U+FFFD - ignore ///< ignore invalid UTF-8 sequences + ignore, ///< ignore invalid UTF-8 sequences + keep ///< keep invalid UTF-8 sequences; their bytes are copied unchanged }; template @@ -1019,6 +1020,47 @@ class serializer break; } + case error_handler_t::keep: + { + // drop whatever the incomplete sequence left in + // the buffer (only copied if !EnsureAscii) and copy + // the ill-formed bytes from the input instead + bytes = bytes_after_last_accept; + + if (undumped_chars > 0) + { + // the pending bytes of the incomplete sequence + // are ill-formed; the current byte may be OK for + // itself, so we would like to read it again + for (std::size_t j = i - undumped_chars; j < i; ++j) + { + string_buffer[bytes++] = s[j]; + } + --i; + } + else + { + // the current byte cannot start any sequence + string_buffer[bytes++] = s[i]; + } + + // write buffer and reset index; there must be 13 bytes + // left, as this is the maximal number of bytes to be + // written ("\uxxxx\uxxxx\0") for one code point + if (string_buffer.size() - bytes < 13) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + + bytes_after_last_accept = bytes; + undumped_chars = 0; + + // continue processing the string + state = UTF8_ACCEPT; + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -1064,6 +1106,15 @@ class serializer break; } + case error_handler_t::keep: + { + // write all accepted bytes + put_buffer(string_buffer, bytes_after_last_accept); + // copy the bytes of the incomplete sequence unchanged + put_string(s, s.size() - undumped_chars, s.size()); + break; + } + case error_handler_t::replace: { // write all accepted bytes diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 4981c3b02..cfb007dca 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -23006,7 +23006,8 @@ enum class error_handler_t { strict, ///< throw a type_error exception in case of invalid UTF-8 replace, ///< replace invalid UTF-8 sequences with U+FFFD - ignore ///< ignore invalid UTF-8 sequences + ignore, ///< ignore invalid UTF-8 sequences + keep ///< keep invalid UTF-8 sequences; their bytes are copied unchanged }; template @@ -23977,6 +23978,47 @@ class serializer break; } + case error_handler_t::keep: + { + // drop whatever the incomplete sequence left in + // the buffer (only copied if !EnsureAscii) and copy + // the ill-formed bytes from the input instead + bytes = bytes_after_last_accept; + + if (undumped_chars > 0) + { + // the pending bytes of the incomplete sequence + // are ill-formed; the current byte may be OK for + // itself, so we would like to read it again + for (std::size_t j = i - undumped_chars; j < i; ++j) + { + string_buffer[bytes++] = s[j]; + } + --i; + } + else + { + // the current byte cannot start any sequence + string_buffer[bytes++] = s[i]; + } + + // write buffer and reset index; there must be 13 bytes + // left, as this is the maximal number of bytes to be + // written ("\uxxxx\uxxxx\0") for one code point + if (string_buffer.size() - bytes < 13) + { + put_buffer(string_buffer, bytes); + bytes = 0; + } + + bytes_after_last_accept = bytes; + undumped_chars = 0; + + // continue processing the string + state = UTF8_ACCEPT; + break; + } + default: // LCOV_EXCL_LINE JSON_ASSERT(false); // NOLINT(cert-dcl03-c,hicpp-static-assert,misc-static-assert) LCOV_EXCL_LINE } @@ -24022,6 +24064,15 @@ class serializer break; } + case error_handler_t::keep: + { + // write all accepted bytes + put_buffer(string_buffer, bytes_after_last_accept); + // copy the bytes of the incomplete sequence unchanged + put_string(s, s.size() - undumped_chars, s.size()); + break; + } + case error_handler_t::replace: { // write all accepted bytes diff --git a/tests/src/unit-regression2.cpp b/tests/src/unit-regression2.cpp index b128b7a73..4a65ad3b2 100644 --- a/tests/src/unit-regression2.cpp +++ b/tests/src/unit-regression2.cpp @@ -765,6 +765,15 @@ TEST_CASE("regression tests 2") CHECK(j == k); } + SECTION("issue #4552 - UTF-8 invalid characters are not always ignored when dumping with error_handler_t::ignore") + { + json node; + node["test"] = "test\334\005"; + CHECK(node.dump(-1, ' ', false, json::error_handler_t::ignore) == "{\"test\":\"test\\u0005\"}"); + CHECK(node.dump(-1, ' ', false, json::error_handler_t::keep) == "{\"test\":\"test\334\\u0005\"}"); + CHECK(node.dump(-1, ' ', true, json::error_handler_t::keep) == "{\"test\":\"test\334\\u0005\"}"); + } + } TEST_CASE("regression test - parser callback must not lose a duplicate key's prior value") diff --git a/tests/src/unit-serialization.cpp b/tests/src/unit-serialization.cpp index 45617c3b3..a9c9ffcf3 100644 --- a/tests/src/unit-serialization.cpp +++ b/tests/src/unit-serialization.cpp @@ -92,6 +92,8 @@ TEST_CASE("serialization") CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"äü\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"ä\xEF\xBF\xBDü\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"\\u00e4\\ufffd\\u00fc\""); + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"ä\xA9ü\""); + CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"\\u00e4\xA9\\u00fc\""); } SECTION("invalid character (regression guard for shared UTF-8 decoder, see #5529)") @@ -114,6 +116,8 @@ TEST_CASE("serialization") CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd\""); + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"123\xC2\""); + CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"123\xC2\""); } SECTION("unexpected character") @@ -126,6 +130,39 @@ TEST_CASE("serialization") CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123456\""); CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\x34\x35\x36\""); CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd456\""); + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"123\xF1\xB0\x34\x35\x36\""); + CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"123\xF1\xB0\x34\x35\x36\""); + } + + SECTION("keep: valid characters are still escaped") + { + // an invalid byte followed by characters that must be escaped + const json j = "\xC2\"\\\n\xFF\x05"; + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xC2\\\"\\\\\\n\xFF\\u0005\""); + CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xC2\\\"\\\\\\n\xFF\\u0005\""); + } + + SECTION("keep: truncated multibyte sequences") + { + CHECK(json("\xF0\x9F\x98").dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xF0\x9F\x98\""); + CHECK(json("\xF0\x9F\x98").dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xF0\x9F\x98\""); + CHECK(json("\xF0\x9F\x98" "a").dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xF0\x9F\x98" "a\""); + CHECK(json("\xF0\x9F\x98" "a").dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xF0\x9F\x98" "a\""); + } + + SECTION("keep: long string with many invalid bytes") + { + // exceeds the internal string buffer several times + std::string input; + std::string expected = "\""; + for (int i = 0; i < 2000; ++i) + { + input += "\xFF\xE2\x82\n\xC3\xA4"; + expected += "\xFF\xE2\x82\\n\xC3\xA4"; + } + expected += "\""; + const json j = input; + CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == expected); } SECTION("U+FFFD Substitution of Maximal Subparts") diff --git a/tests/src/unit-unicode2.cpp b/tests/src/unit-unicode2.cpp index a9649b4de..f77528d5a 100644 --- a/tests/src/unit-unicode2.cpp +++ b/tests/src/unit-unicode2.cpp @@ -14,6 +14,7 @@ #include using nlohmann::json; +#include #include #include #include @@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 static std::string s_replaced2; static std::string s_replaced_ascii; static std::string s_replaced2_ascii; + static std::string s_kept; + static std::string s_kept2; + static std::string s_kept_ascii; - // dumping with ignore/replace must not throw in any case + // dumping with ignore/replace/keep must not throw in any case s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore); @@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace); s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace); s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace); + s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep); if (success_expected) { @@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // all dumps should agree on the string CHECK(s_strict == s_ignored); CHECK(s_strict == s_replaced); + CHECK(s_strict == s_kept); } else { @@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // check that replace string contains a replacement character CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos); + + // ignore drops the invalid bytes, keep copies them + CHECK(s_ignored != s_kept); + CHECK(s_ignored_ascii != s_kept_ascii); + + // unless a byte needs escaping, keep copies the input unchanged + const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c) + { + return static_cast(c) < 0x20 || c == '"' || c == '\\'; + }); + if (!needs_escaping) + { + CHECK(s_kept == "\"" + json_string + "\""); + } } // check that prefix and suffix are preserved @@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz"); CHECK(s_replaced2_ascii.substr(1, 3) == "abc"); CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz"); + CHECK(s_kept2.substr(1, 3) == "abc"); + CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz"); } void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4); diff --git a/tests/src/unit-unicode3.cpp b/tests/src/unit-unicode3.cpp index 12c12eea4..e4877a0ba 100644 --- a/tests/src/unit-unicode3.cpp +++ b/tests/src/unit-unicode3.cpp @@ -14,6 +14,7 @@ #include using nlohmann::json; +#include #include #include #include @@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 static std::string s_replaced2; static std::string s_replaced_ascii; static std::string s_replaced2_ascii; + static std::string s_kept; + static std::string s_kept2; + static std::string s_kept_ascii; - // dumping with ignore/replace must not throw in any case + // dumping with ignore/replace/keep must not throw in any case s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore); @@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace); s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace); s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace); + s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep); if (success_expected) { @@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // all dumps should agree on the string CHECK(s_strict == s_ignored); CHECK(s_strict == s_replaced); + CHECK(s_strict == s_kept); } else { @@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // check that replace string contains a replacement character CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos); + + // ignore drops the invalid bytes, keep copies them + CHECK(s_ignored != s_kept); + CHECK(s_ignored_ascii != s_kept_ascii); + + // unless a byte needs escaping, keep copies the input unchanged + const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c) + { + return static_cast(c) < 0x20 || c == '"' || c == '\\'; + }); + if (!needs_escaping) + { + CHECK(s_kept == "\"" + json_string + "\""); + } } // check that prefix and suffix are preserved @@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz"); CHECK(s_replaced2_ascii.substr(1, 3) == "abc"); CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz"); + CHECK(s_kept2.substr(1, 3) == "abc"); + CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz"); } void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4); diff --git a/tests/src/unit-unicode4.cpp b/tests/src/unit-unicode4.cpp index 43cf7095e..4265a7e84 100644 --- a/tests/src/unit-unicode4.cpp +++ b/tests/src/unit-unicode4.cpp @@ -14,6 +14,7 @@ #include using nlohmann::json; +#include #include #include #include @@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 static std::string s_replaced2; static std::string s_replaced_ascii; static std::string s_replaced2_ascii; + static std::string s_kept; + static std::string s_kept2; + static std::string s_kept_ascii; - // dumping with ignore/replace must not throw in any case + // dumping with ignore/replace/keep must not throw in any case s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore); @@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace); s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace); s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace); + s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep); if (success_expected) { @@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // all dumps should agree on the string CHECK(s_strict == s_ignored); CHECK(s_strict == s_replaced); + CHECK(s_strict == s_kept); } else { @@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // check that replace string contains a replacement character CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos); + + // ignore drops the invalid bytes, keep copies them + CHECK(s_ignored != s_kept); + CHECK(s_ignored_ascii != s_kept_ascii); + + // unless a byte needs escaping, keep copies the input unchanged + const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c) + { + return static_cast(c) < 0x20 || c == '"' || c == '\\'; + }); + if (!needs_escaping) + { + CHECK(s_kept == "\"" + json_string + "\""); + } } // check that prefix and suffix are preserved @@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz"); CHECK(s_replaced2_ascii.substr(1, 3) == "abc"); CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz"); + CHECK(s_kept2.substr(1, 3) == "abc"); + CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz"); } void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4); diff --git a/tests/src/unit-unicode5.cpp b/tests/src/unit-unicode5.cpp index bc0312820..8d45b7b85 100644 --- a/tests/src/unit-unicode5.cpp +++ b/tests/src/unit-unicode5.cpp @@ -14,6 +14,7 @@ #include using nlohmann::json; +#include #include #include #include @@ -75,8 +76,11 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 static std::string s_replaced2; static std::string s_replaced_ascii; static std::string s_replaced2_ascii; + static std::string s_kept; + static std::string s_kept2; + static std::string s_kept_ascii; - // dumping with ignore/replace must not throw in any case + // dumping with ignore/replace/keep must not throw in any case s_ignored = j.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored2 = j2.dump(-1, ' ', false, json::error_handler_t::ignore); s_ignored_ascii = j.dump(-1, ' ', true, json::error_handler_t::ignore); @@ -85,6 +89,9 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 s_replaced2 = j2.dump(-1, ' ', false, json::error_handler_t::replace); s_replaced_ascii = j.dump(-1, ' ', true, json::error_handler_t::replace); s_replaced2_ascii = j2.dump(-1, ' ', true, json::error_handler_t::replace); + s_kept = j.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept2 = j2.dump(-1, ' ', false, json::error_handler_t::keep); + s_kept_ascii = j.dump(-1, ' ', true, json::error_handler_t::keep); if (success_expected) { @@ -94,6 +101,7 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // all dumps should agree on the string CHECK(s_strict == s_ignored); CHECK(s_strict == s_replaced); + CHECK(s_strict == s_kept); } else { @@ -105,6 +113,20 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 // check that replace string contains a replacement character CHECK(s_replaced.find("\xEF\xBF\xBD") != std::string::npos); + + // ignore drops the invalid bytes, keep copies them + CHECK(s_ignored != s_kept); + CHECK(s_ignored_ascii != s_kept_ascii); + + // unless a byte needs escaping, keep copies the input unchanged + const bool needs_escaping = std::any_of(json_string.begin(), json_string.end(), [](char c) + { + return static_cast(c) < 0x20 || c == '"' || c == '\\'; + }); + if (!needs_escaping) + { + CHECK(s_kept == "\"" + json_string + "\""); + } } // check that prefix and suffix are preserved @@ -116,6 +138,8 @@ void check_utf8dump(bool success_expected, int byte1, int byte2 = -1, int byte3 CHECK(s_replaced2.substr(s_replaced2.size() - 4, 3) == "xyz"); CHECK(s_replaced2_ascii.substr(1, 3) == "abc"); CHECK(s_replaced2_ascii.substr(s_replaced2_ascii.size() - 4, 3) == "xyz"); + CHECK(s_kept2.substr(1, 3) == "abc"); + CHECK(s_kept2.substr(s_kept2.size() - 4, 3) == "xyz"); } void check_utf8string(bool success_expected, int byte1, int byte2, int byte3, int byte4);