diff --git a/.github/labeler.yml b/.github/labeler.yml index 5d0c53e41..bc1621a06 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -51,7 +51,7 @@ labels: - "include/nlohmann/detail/view/.*" - "single_include/nlohmann/json_view\\.hpp" - "tests/src/unit-json_view.*" - - "tests/src/fuzzer-parse_json_view\\.cpp" + - "tests/src/fuzzer-(parse_json_view|json_view_image)\\.cpp" - "tests/benchmarks/json_view/.*" - "tools/amalgamate/config_json_view\\.json" - "docs/mkdocs/docs/features/json_view\\.md" diff --git a/BUILD.bazel b/BUILD.bazel index 273d93762..c7ceba036 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -72,6 +72,7 @@ cc_library( "include/nlohmann/detail/view/edit.hpp", "include/nlohmann/detail/view/edit_storage.hpp", "include/nlohmann/detail/view/errors.hpp", + "include/nlohmann/detail/view/image.hpp", "include/nlohmann/detail/view/input.hpp", "include/nlohmann/detail/view/iterator.hpp", "include/nlohmann/detail/view/lookup.hpp", diff --git a/include/nlohmann/detail/view/document_data.hpp b/include/nlohmann/detail/view/document_data.hpp index 9b3842e2c..6ba3c78ee 100644 --- a/include/nlohmann/detail/view/document_data.hpp +++ b/include/nlohmann/detail/view/document_data.hpp @@ -10,7 +10,7 @@ #include // array #include // size_t -#include // uint32_t +#include // uint8_t, uint32_t #include // memcpy #include // less #include // map @@ -41,7 +41,9 @@ struct document_data node* inline_tape = nullptr; ///< node array allocated together with this header std::size_t inline_cap = 0; std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init) + std::size_t arena_size = 0; ///< bytes of decoded strings at base[1] (the arena, or those of a loaded image) std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init) + std::vector owned_image{}; ///< a loaded image the document owns (the text and the decoded strings point into it) // NOLINT(readability-redundant-member-init) // hash indexes of large objects (see object_index.hpp) static constexpr std::uint32_t index_min_members = 128; diff --git a/include/nlohmann/detail/view/image.hpp b/include/nlohmann/detail/view/image.hpp new file mode 100644 index 000000000..e7de241ef --- /dev/null +++ b/include/nlohmann/detail/view/image.hpp @@ -0,0 +1,602 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // size_t +#include // uint8_t, uint16_t, uint32_t, uint64_t +#include // memcmp, memcpy +#include // numeric_limits +#include // string +#include // vector + +#include +#include +#include +#include +#include +#include +#include +#include + +// Images: a document stored so that loading it needs no parsing. +// +// Layout (little-endian): a 64-byte header, the nodes, the text (the source, +// followed by the number tokens written by edits), a NUL, the decoded strings +// (followed by the strings written by edits), a NUL. The idea is that of +// zero-copy formats such as FlatBuffers (https://github.com/google/flatbuffers) +// and YaFF (https://github.com/yandex/yaff); no code is taken from them. +// check_image follows the idea of FlatBuffers' Verifier (bounds and +// structure) and also checks what the parser guarantees about strings and +// numbers, so that reading and serializing a checked image is safe and yields +// valid JSON. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// how load() checks an image +enum class image_check +{ + /// everything the parser guarantees: structure and bounds, strings (valid + /// UTF-8; source strings without quotes, backslashes, and control + /// characters), and numbers (well-formed, matching the stored values) + full, + /// structure and bounds only: reading and serializing are safe, but a + /// crafted image can yield invalid UTF-8, strings that serialize to + /// invalid JSON, or numbers that differ from their text + bounds, + /// none: for images from a trusted source only (a damaged image is + /// undefined behavior) + none, +}; + +struct image_header +{ + std::array magic; ///< "NJVI" + std::uint32_t version; ///< 1 + std::uint64_t node_count; + std::uint64_t text_size; + std::uint64_t arena_size; + std::array reserved; ///< zero (for later versions) +}; +static_assert(sizeof(image_header) == 64, "the image header must be 64 bytes"); + +constexpr std::uint32_t image_version = 1; + +/// the largest node count and text or string size of an image (as for parsed +/// documents, offsets and counts must fit 32 bits) +constexpr std::uint64_t image_limit = 0xFFFFFFF0u; + +/// Copy the current structure of an edited document into nodes in document +/// order, as the parser would have written them. Text written by edits is +/// appended to text_tail (number tokens) and arena_tail (strings); floats that +/// are not finite become null, as dump() writes them. +inline void compact_nodes(const document_data& d, std::size_t arena_size, std::vector& out, std::string& text_tail, std::string& arena_tail) +{ + struct frame + { + const node* cur; + const node* end; + std::size_t index; ///< the container's node in out + std::uint32_t count; + bool object; + }; + std::vector stack; + const auto string_node = [&](const node & s) + { + node r = s; + r.extra = 0; + r.flags = static_cast(s.flags & node_flags::storage); + if (r.flags == node_flags::edited) + { + r.off = static_cast(arena_size + arena_tail.size()); + arena_tail.append(d.str(s), s.len); + r.flags = node_flags::escaped; + } + return r; + }; + const auto emit = [&](const node * v) + { + node r = *v; + switch (static_cast(v->kind)) + { + case value_t::object: + case value_t::array: + r.flags = 0; + r.extra = 0; + r.off = (v->flags & (node_flags::moved | node_flags::is_new)) != 0 ? 0 : v->off; + r.len = 0; // counted below + r.next = 0; // set when the container is complete + stack.push_back(frame{d.first_child_edited(v), d.child_end_edited(v), out.size(), 0, v->kind == static_cast(value_t::object)}); + break; + case value_t::string: + r = string_node(*v); + break; + case value_t::number_integer: + case value_t::number_unsigned: + if ((v->flags & node_flags::storage) == node_flags::edited) + { + r.off = static_cast(d.size + text_tail.size()); + text_tail.append(d.str(*v), number_length(*v)); + } + r.flags = 0; + break; + case value_t::number_float: + if ((v->flags & node_flags::storage) == node_flags::edited) + { + const char* const t = d.str(*v); + if (t[0] == 'n' || t[0] == 'i' || (v->len > 1 && t[1] == 'i')) + { + r = node{}; // nan and infinity: null, as dump() writes them + r.kind = static_cast(value_t::null); + break; + } + r.off = static_cast(d.size + text_tail.size()); + text_tail.append(t, v->len); + r.extra = 0xFFFFu; // the digit layout is not recorded + } + r.flags = 0; + break; + case value_t::boolean: + r.flags = static_cast(v->flags & node_flags::is_true); + break; + case value_t::null: + case value_t::binary: + case value_t::discarded: + default: + r.flags = 0; + break; + } + out.push_back(r); + }; + emit(d.tape); + while (!stack.empty()) + { + frame& top = stack.back(); + if (top.cur == top.end) + { + node& c = out[top.index]; + c.len = top.count; + c.next = static_cast(out.size() - top.index); + stack.pop_back(); + continue; + } + ++top.count; + const node* v = nullptr; + if (top.object) + { + out.push_back(string_node(*top.cur)); + v = document_data::deref(top.cur + 1); + top.cur = document_data::after(top.cur + 1); + } + else + { + v = document_data::deref(top.cur); + top.cur = document_data::after(top.cur); + } + emit(v); // may grow the stack (top is not used afterwards) + } +} + +/// the document as an image +inline std::vector save_image(const document_data& d) +{ +#if !NLOHMANN_VIEW_LITTLE_ENDIAN + throw_type_error(320, "json_document images need a little-endian target"); // LCOV_EXCL_LINE +#endif + const std::size_t arena_size = d.arena_size; + const node* nodes = d.tape; + std::size_t count = d.tape_size; + std::vector compacted; + std::string text_tail; + std::string arena_tail; + if (d.edits) + { + compact_nodes(d, arena_size, compacted, text_tail, arena_tail); + nodes = compacted.data(); + count = compacted.size(); + } + const std::size_t text_size = d.size + text_tail.size(); + const std::size_t total_arena = arena_size + arena_tail.size(); + if (NLOHMANN_VIEW_UNLIKELY(text_size >= image_limit || total_arena >= image_limit || count >= image_limit)) + { + // LCOV_EXCL_START (4 GiB) + throw_out_of_range(416, "images of 4 GiB or more are not supported by json_document"); + // LCOV_EXCL_STOP + } + image_header h{}; + h.magic = {{'N', 'J', 'V', 'I'}}; + h.version = image_version; + h.node_count = count; + h.text_size = text_size; + h.arena_size = total_arena; + std::vector image(sizeof(h) + (count * sizeof(node)) + text_size + 1 + total_arena + 1); + std::uint8_t* o = image.data(); + std::memcpy(o, &h, sizeof(h)); + o += sizeof(h); + std::memcpy(o, nodes, count * sizeof(node)); + // the hash indexes are rebuilt by load() + for (std::size_t i = 0; i < count; ++i) + { + if (nodes[i].kind == static_cast(value_t::object) && nodes[i].extra != 0) + { + node n = nodes[i]; + n.extra = 0; + std::memcpy(o + (i * sizeof(node)), &n, sizeof(node)); + } + } + o += count * sizeof(node); + const auto append = [&o](const char* s, std::size_t n) + { + if (n != 0) + { + std::memcpy(o, s, n); + o += n; + } + }; + append(d.src, d.size); + append(text_tail.data(), text_tail.size()); + *o++ = 0; + append(d.base[1], arena_size); + append(arena_tail.data(), arena_tail.size()); + *o = 0; + return image; +} + +/// whether a number node matches its token the way the parser records it +/// (after the bounds check) +inline bool check_number(const node& n, const unsigned char* text) +{ + const std::size_t len = number_length(n); + const unsigned char* const s = text + n.off; + const unsigned char* const e = s + len; + const unsigned char* p = s; + const bool negative = *p == '-'; + p += negative ? 1 : 0; + const unsigned char* const int_start = p; + if (p == e) + { + return false; + } + if (*p == '0') + { + ++p; + } + else if (*p >= '1' && *p <= '9') + { + while (p != e && is_digit(*p)) + { + ++p; + } + } + else + { + return false; + } + const auto int_digits = static_cast(p - int_start); + std::size_t frac_digits = 0; + bool is_float = false; + if (p != e && *p == '.') + { + const unsigned char* const f0 = ++p; + while (p != e && is_digit(*p)) + { + ++p; + } + if (p == f0) + { + return false; + } + frac_digits = static_cast(p - f0); + is_float = true; + } + long exponent = 0; + if (p != e && (*p | 0x20u) == 'e') + { + ++p; + const bool exp_negative = p != e && *p == '-'; + p += (p != e && (*p == '+' || *p == '-')) ? 1 : 0; + if (p == e || !is_digit(*p)) + { + return false; + } + while (p != e && is_digit(*p)) + { + exponent = exponent < 100000 ? (exponent * 10) + (*p - '0') : exponent; + ++p; + } + exponent = exp_negative ? -exponent : exponent; + is_float = true; + } + if (p != e) + { + return false; + } + if (n.kind == static_cast(value_t::number_float)) + { + // the digit layout the parser records (or "many", as compaction + // writes it), and a finite value + const auto layout = static_cast((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8u)); + if (n.extra != layout && n.extra != 0xFFFFu) + { + return false; + } + // parse() rejects floats that overflow; as there, only a number whose + // magnitude could reach 1e308 needs the conversion + if (static_cast(int_digits) + exponent > 300) + { + const auto v = float_value(reinterpret_cast(s), n); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return v <= (std::numeric_limits::max)() && v >= -(std::numeric_limits::max)(); + } + return true; + } + // integers: the token's value is the stored one; number_integer nodes of + // edits can be non-negative (as basic_json keeps the type of a value) + const bool integer = n.kind == static_cast(value_t::number_integer); + if (is_float || int_digits > 20 || (negative && !integer)) + { + return false; + } + // (at most 19 digits cannot overflow; 20 digits are compared with 2^64 - 1) + if (int_digits == 20 && std::memcmp(int_start, "18446744073709551615", 20) > 0) + { + return false; + } + std::uint64_t m = 0; + for (const unsigned char* d = int_start; d != int_start + int_digits; ++d) + { + m = (m * 10) + static_cast(*d - '0'); + } + if (integer && m > (negative ? std::uint64_t{1} << 63u : (std::uint64_t{1} << 63u) - 1)) + { + return false; + } + return integer_bits(n) == (negative ? 0 - m : m); +} + +/// Check the nodes of a loaded image against its text and decoded strings: +/// kinds, flags, and `extra`; extents and element counts of arrays and +/// objects; keys; bounds; string contents (source strings as the parser +/// leaves them: no quotes, backslashes, or control characters; all strings +/// valid UTF-8); and number tokens. +inline bool check_image(const node* nodes, std::size_t count, const unsigned char* text, std::size_t text_size, + const unsigned char* arena, std::size_t arena_size, bool full) +{ + struct frame + { + std::size_t end; + std::uint32_t len; + std::uint32_t seen; + bool object; + bool expect_key; + }; + std::vector stack; + const auto check_string = [&](const node & n) -> bool + { + if ((n.flags & ~node_flags::escaped) != 0 || n.extra != 0) + { + return false; + } + const bool decoded = (n.flags & node_flags::escaped) != 0; + const unsigned char* const base = decoded ? arena : text; + const std::size_t limit = decoded ? arena_size : text_size; + if (n.off > limit || n.len > limit - n.off) + { + return false; + } + if (!full) + { + return true; + } + const unsigned char* const b = base + n.off; + return decoded ? valid_utf8_prefix(b, n.len) == n.len : scan_string_run(b, b + n.len) == b + n.len; + }; + // bounds of a number token; the recorded digit layout must lie within it + const auto number_in_bounds = [&](const node & n) -> bool + { + const std::size_t len = number_length(n); + if (len == 0 || n.off > text_size || len > text_size - n.off) + { + return false; + } + if (n.kind != static_cast(value_t::number_float)) + { + return (n.extra >> 8u) == 0; + } + // float_value() reads the sign, the integer digits, and the point and + // fraction digits the layout records (a layout of more than 19 digits + // means the general conversion, which stays within the token) + const std::size_t int_digits = n.extra & 0xFFu; + const std::size_t frac_digits = n.extra >> 8u; + const std::size_t need = (text[n.off] == '-' ? 1u : 0u) + int_digits + (frac_digits != 0 ? frac_digits + 1 : 0); + return int_digits + frac_digits > 19 || need <= len; + }; + std::size_t i = 0; + for (;;) + { + // close finished arrays and objects + while (!stack.empty() && i == stack.back().end) + { + const frame f = stack.back(); + if (f.seen != f.len || (f.object && !f.expect_key)) + { + return false; + } + stack.pop_back(); + if (!stack.empty()) + { + ++stack.back().seen; + stack.back().expect_key = true; + } + } + if (i == count) + { + return stack.empty(); + } + if (i != 0 && stack.empty()) + { + return false; // nodes after the root + } + const node& n = nodes[i]; + if (!stack.empty() && stack.back().object && stack.back().expect_key) + { + if (n.kind != static_cast(value_t::string) || !check_string(n)) + { + return false; + } + stack.back().expect_key = false; + ++i; + continue; + } + bool complete = true; + switch (static_cast(n.kind)) + { + case value_t::null: + // (the offset of a literal is read to size the output of dump()) + if (n.flags != 0 || n.extra != 0 || n.off > text_size) + { + return false; + } + break; + case value_t::boolean: + if ((n.flags & ~node_flags::is_true) != 0 || n.extra != 0 || n.off > text_size) + { + return false; + } + break; + case value_t::string: + if (!check_string(n)) + { + return false; + } + break; + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + if (n.flags != 0 || !number_in_bounds(n) || (full && !check_number(n, text))) + { + return false; + } + break; + case value_t::array: + case value_t::object: + { + const std::size_t limit = stack.empty() ? count : stack.back().end; + if (n.flags != 0 || n.extra != 0 || n.next == 0 || n.next > limit - i || n.off > text_size) + { + return false; + } + stack.push_back(frame{i + n.next, n.len, 0, n.kind == static_cast(value_t::object), true}); + complete = false; + break; + } + case value_t::binary: + case value_t::discarded: + default: + return false; + } + ++i; + if (complete && !stack.empty()) + { + ++stack.back().seen; + stack.back().expect_key = true; + } + } +} + +[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_invalid_image(const char* what) +{ + throw_parse_error(116, concat("invalid json_document image: ", what)); +} + +/// Read an image into d. The text and the decoded strings stay in the image; +/// the nodes are copied (so that they are aligned, and edits can change them). +inline void load_image(document_data& d, const std::uint8_t* image, std::size_t size, image_check check) +{ +#if !NLOHMANN_VIEW_LITTLE_ENDIAN + throw_type_error(320, "json_document images need a little-endian target"); // LCOV_EXCL_LINE +#endif + if (image == nullptr || size < sizeof(image_header)) + { + throw_invalid_image("too short"); + } + image_header h{}; + std::memcpy(&h, image, sizeof(h)); + // (the reserved fields are for later versions) + if (std::memcmp(h.magic.data(), "NJVI", 4) != 0 || h.version != image_version + || (h.reserved[0] | h.reserved[1] | h.reserved[2] | h.reserved[3]) != 0) + { + throw_invalid_image("unknown format"); + } + const std::size_t room = size - sizeof(h); + if (h.node_count == 0 || h.node_count > room / sizeof(node) || h.node_count >= image_limit || h.text_size >= image_limit || h.arena_size >= image_limit) + { + throw_invalid_image("sizes out of range"); + } + const auto count = static_cast(h.node_count); + const auto text_size = static_cast(h.text_size); + const auto arena_size = static_cast(h.arena_size); + const std::size_t text_at = sizeof(h) + (count * sizeof(node)); + // the text, a NUL, the decoded strings, a NUL, and nothing after them + if (size - text_at < 2 || text_size > size - text_at - 2 || arena_size != size - text_at - text_size - 2 + || image[text_at + text_size] != 0 || image[size - 1] != 0) + { + throw_invalid_image("sizes out of range"); + } + + d.discarded = true; + d.edits.reset(); + d.base[2] = nullptr; + d.owned.clear(); + if (d.owned_image.empty() || image != d.owned_image.data()) + { + d.owned_image.clear(); + } + d.arena.clear(); + d.indexes.clear(); + d.index_slots.clear(); + d.large_objects.clear(); + d.tape_size = 0; + d.reserve(count); + std::memcpy(d.tape, image + sizeof(h), count * sizeof(node)); + d.tape_size = count; + const std::uint8_t* const text = image + text_at; + const std::uint8_t* const arena = text + text_size + 1; + d.src = reinterpret_cast(text); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + d.size = text_size; + d.base[0] = d.src; + d.base[1] = reinterpret_cast(arena); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + d.arena_size = arena_size; + if (check != image_check::none && !check_image(d.tape, count, text, text_size, arena, arena_size, check == image_check::full)) + { + throw_invalid_image("the check failed"); + } + // the hash indexes of large objects, as after parsing + for (std::size_t i = 0; i < count; ++i) + { + node& n = d.tape[i]; + if (n.kind == static_cast(value_t::object)) + { + n.extra = 0; + if (n.len >= document_data::index_min_members) + { + d.large_objects.push_back(static_cast(i)); + } + } + } + build_object_indexes(d); + d.discarded = false; +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/number.hpp b/include/nlohmann/detail/view/number.hpp index fce98e9ba..81e3d2b1e 100644 --- a/include/nlohmann/detail/view/number.hpp +++ b/include/nlohmann/detail/view/number.hpp @@ -26,43 +26,76 @@ namespace detail namespace view { +/*! +@brief locate the decimal point and the end of the mantissa of a float token + +Also checks that the token is a JSON number. Tokens of the parser and of edits +always are; an image loaded with image_check::bounds can hold any bytes, which +must not reach the conversion (it expects a well-formed token). +*/ +inline bool float_token_layout(const char* first, const char* last, std::size_t& dot, std::size_t& mantissa_end) noexcept +{ + const auto digit = [last](const char* q) + { + return q != last && is_digit(static_cast(*q)); + }; + const char* p = first; + p += (p != last && *p == '-') ? 1 : 0; + if (!digit(p) || (*p == '0' && digit(p + 1))) + { + return false; + } + while (digit(p)) + { + ++p; + } + dot = std::string::npos; + if (p != last && *p == '.') + { + dot = static_cast(p - first); + if (!digit(++p)) + { + return false; + } + while (digit(p)) + { + ++p; + } + } + mantissa_end = static_cast(p - first); + if (p != last && (*p == 'e' || *p == 'E')) + { + ++p; + p += (p != last && (*p == '+' || *p == '-')) ? 1 : 0; + if (!digit(p)) + { + return false; + } + while (digit(p)) + { + ++p; + } + } + return p == last; +} + /*! @brief the value of the float token of a node, as parse() converts it Uses the lexer's conversion (detail::convert_float), so that the values are bit-identical to parse(): float and double are converted without allocation -and independent of the locale. The digit layout recorded while parsing locates -the decimal point and the exponent without scanning the token. +and independent of the locale. A token that is not a JSON number (only in a +damaged image loaded with image_check::bounds) yields 0. */ template NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) { const char* const last = first + n.len; - const std::size_t neg = first[0] == '-' ? 1 : 0; - const std::size_t int_digits = n.extra & 0xFFu; - const std::size_t frac_digits = n.extra >> 8u; - std::size_t dot = std::string::npos; - std::size_t mantissa_end = n.len; - if (int_digits != 255 && frac_digits != 255) + std::size_t dot = 0; + std::size_t mantissa_end = 0; + if (NLOHMANN_VIEW_UNLIKELY(!float_token_layout(first, last, dot, mantissa_end))) { - dot = frac_digits != 0 ? neg + int_digits : std::string::npos; - mantissa_end = neg + int_digits + (frac_digits != 0 ? 1 + frac_digits : 0); - } - else - { - // more digits than the layout records: locate them - for (std::size_t i = 0; i < n.len; ++i) - { - if (first[i] == '.') - { - dot = i; - } - else if (first[i] == 'e' || first[i] == 'E') - { - mantissa_end = i; - break; - } - } + return FloatType{}; } return convert_float(first, last, dot, mantissa_end); } @@ -93,16 +126,20 @@ NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char } if (p != e) { - // [eE][+-]digits; huge exponents saturate (the parser rejected overflow) + // [eE][+-]digits; huge exponents saturate (the parser rejected + // overflow). The token is not read beyond e, and the digits are taken + // as unsigned, so that a token that is not well-formed (a damaged + // image loaded with image_check::bounds) yields a wrong value, but no + // overflow. ++p; - const bool exp_negative = *p == '-'; - p += (*p == '-' || *p == '+') ? 1 : 0; + const bool exp_negative = p != e && *p == '-'; + p += (p != e && (*p == '-' || *p == '+')) ? 1 : 0; std::int64_t exp_value = 0; for (; p != e; ++p) { if (exp_value < 0x10000000) { - exp_value = (exp_value * 10) + (*p - '0'); + exp_value = (exp_value * 10) + static_cast(*p - '0'); } } q += exp_negative ? -exp_value : exp_value; diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index 3c5addf11..64d0bd933 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -317,7 +317,9 @@ class view_serializer m_out.put('"'); } - /// as serializer::dump_escaped() for valid UTF-8 (the view has no other) + /// as serializer::dump_escaped(); strings of a document are valid UTF-8, + /// except in a damaged image loaded with image_check::bounds, for which + /// this throws what basic_json::dump() throws for the string template void write_escaped(const unsigned char* s, std::size_t n) { @@ -341,12 +343,13 @@ class view_serializer } std::uint32_t codepoint = s[i]; std::size_t len = 1; - if (codepoint >= 0xC0) + if (codepoint >= 0x80) { - len = 2; - if (codepoint >= 0xE0) + len = validate_one_utf8(s + i, n - i); + if (NLOHMANN_VIEW_UNLIKELY(len == 0)) { - len = codepoint >= 0xF0 ? 4 : 3; + invalid_utf8(s, n); + return; } codepoint &= 0xFFu >> (len + 1); for (std::size_t k = 1; k < len; ++k) @@ -359,6 +362,13 @@ class view_serializer } } + /// throw what basic_json::dump() throws for a string that is not valid UTF-8 + NLOHMANN_VIEW_NOINLINE static void invalid_utf8(const unsigned char* s, std::size_t n) + { + const string_t dumped = BasicJsonType(string_t(reinterpret_cast(s), n)).dump(); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + static_cast(dumped); + } + template void write_codepoint(std::uint32_t codepoint, const unsigned char* bytes, std::size_t len) { diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 4a6b24832..10352058f 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -25,7 +25,7 @@ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #include // size_t -#include // uint32_t +#include // uint8_t, uint32_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits #include // map @@ -53,6 +53,7 @@ #include #include #include +#include #include #include #include @@ -933,7 +934,7 @@ class basic_json_document /// whether the document holds its own copy of the text bool owns_source() const noexcept { - return m_data && !m_data->owned.empty() && m_data->src == m_data->owned.data(); + return m_data && ((!m_data->owned.empty() && m_data->src == m_data->owned.data()) || !m_data->owned_image.empty()); } /// number of index nodes (values plus object keys) @@ -942,7 +943,7 @@ class basic_json_document return m_data ? m_data->tape_size : 0; } - /// bytes held by the document (index, decoded strings, owned text) + /// bytes held by the document (index, decoded strings, owned text or image) std::size_t memory_usage() const noexcept { if (!m_data) @@ -951,7 +952,7 @@ class basic_json_document } return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node)) + (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0) - + m_data->arena.capacity() + m_data->owned.capacity() + + m_data->arena.capacity() + m_data->owned.capacity() + m_data->owned_image.capacity() + (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t)) + (m_data->large_objects.capacity() * sizeof(std::uint32_t)) + (m_data->edits != nullptr ? m_data->edits->bytes : 0); @@ -971,8 +972,10 @@ class basic_json_document // allocate everything first, so that an exception leaves the document // unchanged + // (the decoded strings of a loaded image stay in the image) + const bool arena_in_use = d.base[1] == d.arena.data(); const bool shrink_arena = d.arena.capacity() > d.arena.size(); - std::string arena(shrink_arena ? d.arena : std::string()); + std::string arena(shrink_arena && arena_in_use ? d.arena : std::string()); // (edits link to the nodes of the index, which then stays in place) const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap && d.edits == nullptr; const bool into_header = d.tape_size <= d.inline_cap; @@ -988,10 +991,62 @@ class basic_json_document if (shrink_arena) { d.arena.swap(arena); - d.base[1] = d.arena.data(); + if (arena_in_use) + { + d.base[1] = d.arena.data(); + } } } + //////////// + // images // + //////////// + + /// how load() checks an image (full, bounds, or none) + using image_check = detail::view::image_check; + + /// The document as an image that load() reads without parsing: the node + /// index, the text, and the decoded strings. An edited document is + /// written in its current state (floats that are not finite become null, + /// as in dump()). + std::vector save() const + { + if (NLOHMANN_VIEW_UNLIKELY(!m_data || m_data->discarded)) + { + detail::view::throw_type_error(320, "cannot save a discarded json_document"); + } + return detail::view::save_image(*m_data); + } + + /// Read an image written by save(). The image is borrowed: it must stay + /// alive and unchanged while the document is used. + NLOHMANN_VIEW_NODISCARD + static basic_json_document load(const std::uint8_t* image, std::size_t size, const image_check check = image_check::full) + { + basic_json_document d; + d.ensure_data(nullptr, 0); + detail::view::load_image(*d.m_data, image, size, check); + return d; + } + + /// read an image (borrowed) + NLOHMANN_VIEW_NODISCARD + static basic_json_document load(const std::vector& image, const image_check check = image_check::full) + { + return load(image.data(), image.size(), check); + } + + /// read an image and keep it (no copy) + NLOHMANN_VIEW_NODISCARD + static basic_json_document load(std::vector&& image, const image_check check = image_check::full) + { + basic_json_document d; + d.ensure_data(nullptr, 0); + d.m_data->owned_image = std::move(image); + detail::view::load_image(*d.m_data, d.m_data->owned_image.data(), d.m_data->owned_image.size(), check); + return d; + } + /////////// // edits // /////////// @@ -1175,6 +1230,7 @@ class basic_json_document { d.owned.clear(); } + d.owned_image.clear(); d.src = src; d.size = size; d.tape_size = 0; @@ -1199,6 +1255,7 @@ class basic_json_document { d.base[0] = d.src; d.base[1] = d.arena.data(); + d.arena_size = d.arena.size(); detail::view::build_object_indexes(d); d.discarded = false; return; diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 7f55ee693..1ca6e7adc 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -25,7 +25,7 @@ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #include // size_t -#include // uint32_t +#include // uint8_t, uint32_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits #include // map @@ -82,7 +82,7 @@ #include // array #include // size_t -#include // uint32_t +#include // uint8_t, uint32_t #include // memcpy #include // less #include // map @@ -306,7 +306,9 @@ struct document_data node* inline_tape = nullptr; ///< node array allocated together with this header std::size_t inline_cap = 0; std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init) + std::size_t arena_size = 0; ///< bytes of decoded strings at base[1] (the arena, or those of a loaded image) std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init) + std::vector owned_image{}; ///< a loaded image the document owns (the text and the decoded strings point into it) // NOLINT(readability-redundant-member-init) // hash indexes of large objects (see object_index.hpp) static constexpr std::uint32_t index_min_members = 128; @@ -3771,6 +3773,840 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // array +#include // size_t +#include // uint8_t, uint16_t, uint32_t, uint64_t +#include // memcmp, memcpy +#include // numeric_limits +#include // string +#include // vector + +// #include +// #include + +// #include + +// #include + +// #include + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t +#include // int64_t, uint64_t +#include // numeric_limits +#include // string +#include // integral_constant + +// #include +// #include + +// #include + +// #include + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/*! +@brief locate the decimal point and the end of the mantissa of a float token + +Also checks that the token is a JSON number. Tokens of the parser and of edits +always are; an image loaded with image_check::bounds can hold any bytes, which +must not reach the conversion (it expects a well-formed token). +*/ +inline bool float_token_layout(const char* first, const char* last, std::size_t& dot, std::size_t& mantissa_end) noexcept +{ + const auto digit = [last](const char* q) + { + return q != last && is_digit(static_cast(*q)); + }; + const char* p = first; + p += (p != last && *p == '-') ? 1 : 0; + if (!digit(p) || (*p == '0' && digit(p + 1))) + { + return false; + } + while (digit(p)) + { + ++p; + } + dot = std::string::npos; + if (p != last && *p == '.') + { + dot = static_cast(p - first); + if (!digit(++p)) + { + return false; + } + while (digit(p)) + { + ++p; + } + } + mantissa_end = static_cast(p - first); + if (p != last && (*p == 'e' || *p == 'E')) + { + ++p; + p += (p != last && (*p == '+' || *p == '-')) ? 1 : 0; + if (!digit(p)) + { + return false; + } + while (digit(p)) + { + ++p; + } + } + return p == last; +} + +/*! +@brief the value of the float token of a node, as parse() converts it + +Uses the lexer's conversion (detail::convert_float), so that the values are +bit-identical to parse(): float and double are converted without allocation +and independent of the locale. A token that is not a JSON number (only in a +damaged image loaded with image_check::bounds) yields 0. +*/ +template +NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) +{ + const char* const last = first + n.len; + std::size_t dot = 0; + std::size_t mantissa_end = 0; + if (NLOHMANN_VIEW_UNLIKELY(!float_token_layout(first, last, dot, mantissa_end))) + { + return FloatType{}; + } + return convert_float(first, last, dot, mantissa_end); +} + +/*! +@brief the digits of a float token with at most 19 digits, from its layout + +The digit layout recorded while parsing says where the integer digits, the +fraction digits, and the exponent are, so the digits are read eight at a +time without scanning. + +@param[in] p first character of the token +@param[in] e end of the token +@param[in] limit end of the readable memory (the source text) +*/ +NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + const bool negative = *p == '-'; + p += negative ? 1 : 0; + std::uint64_t w = parse_upto19(p, int_digits, limit); + p += int_digits; + std::int64_t q = 0; + if (frac_digits != 0) + { + w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit); + p += 1 + frac_digits; + q = -static_cast(frac_digits); + } + if (p != e) + { + // [eE][+-]digits; huge exponents saturate (the parser rejected + // overflow). The token is not read beyond e, and the digits are taken + // as unsigned, so that a token that is not well-formed (a damaged + // image loaded with image_check::bounds) yields a wrong value, but no + // overflow. + ++p; + const bool exp_negative = p != e && *p == '-'; + p += (p != e && (*p == '-' || *p == '+')) ? 1 : 0; + std::int64_t exp_value = 0; + for (; p != e; ++p) + { + if (exp_value < 0x10000000) + { + exp_value = (exp_value * 10) + static_cast(*p - '0'); + } + } + q += exp_negative ? -exp_value : exp_value; + } + + float_significand d; + d.w = w; + d.exponent = q; + d.negative = negative; + return d; +} + +/*! +@brief the value of a float token with at most 19 digits, from its layout + +The result is correctly rounded by the lexer's conversion +(detail::decimal_to_float(): Clinger's fast path where both operands are +exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19 +digits), so it is the value parse() produces. +*/ +template +NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + return decimal_to_float(layout_decimal(p, e, int_digits, frac_digits, limit)); +} + +/// the value of a float set by an edit: its token (the shortest round-trip +/// text, or "nan", "inf", "-inf") in the edit arena +template +NLOHMANN_VIEW_NOINLINE FloatType edited_float(const char* token, const node& n) +{ + if (token[0] == 'n') + { + return std::numeric_limits::quiet_NaN(); + } + if (token[0] == 'i' || (token[0] == '-' && token[1] == 'i')) + { + return token[0] == 'i' ? std::numeric_limits::infinity() : -std::numeric_limits::infinity(); + } + return float_value(token, n); +} + +/// the value of the float token of a node, as parse() converts it; floats and +/// doubles with at most 19 digits are converted from the digit layout +template +FloatType float_value(const document_data& d, const node& n) +{ + if (NLOHMANN_VIEW_UNLIKELY((n.flags & node_flags::storage) == node_flags::edited)) + { + return edited_float(d.str(n), n); + } + return float_value(d, n, std::integral_constant::value> {}); +} + +template +FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/) +{ + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many") + { + // (a float token not written by an edit is in the text) + const auto* const first = reinterpret_cast(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return layout_float(first, first + n.len, int_digits, frac_digits, reinterpret_cast(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return float_value(d.str(n), n); +} + +template +FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/) +{ + return float_value(d.str(n), n); +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +// #include + +// #include + + +// Images: a document stored so that loading it needs no parsing. +// +// Layout (little-endian): a 64-byte header, the nodes, the text (the source, +// followed by the number tokens written by edits), a NUL, the decoded strings +// (followed by the strings written by edits), a NUL. The idea is that of +// zero-copy formats such as FlatBuffers (https://github.com/google/flatbuffers) +// and YaFF (https://github.com/yandex/yaff); no code is taken from them. +// check_image follows the idea of FlatBuffers' Verifier (bounds and +// structure) and also checks what the parser guarantees about strings and +// numbers, so that reading and serializing a checked image is safe and yields +// valid JSON. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// how load() checks an image +enum class image_check +{ + /// everything the parser guarantees: structure and bounds, strings (valid + /// UTF-8; source strings without quotes, backslashes, and control + /// characters), and numbers (well-formed, matching the stored values) + full, + /// structure and bounds only: reading and serializing are safe, but a + /// crafted image can yield invalid UTF-8, strings that serialize to + /// invalid JSON, or numbers that differ from their text + bounds, + /// none: for images from a trusted source only (a damaged image is + /// undefined behavior) + none, +}; + +struct image_header +{ + std::array magic; ///< "NJVI" + std::uint32_t version; ///< 1 + std::uint64_t node_count; + std::uint64_t text_size; + std::uint64_t arena_size; + std::array reserved; ///< zero (for later versions) +}; +static_assert(sizeof(image_header) == 64, "the image header must be 64 bytes"); + +constexpr std::uint32_t image_version = 1; + +/// the largest node count and text or string size of an image (as for parsed +/// documents, offsets and counts must fit 32 bits) +constexpr std::uint64_t image_limit = 0xFFFFFFF0u; + +/// Copy the current structure of an edited document into nodes in document +/// order, as the parser would have written them. Text written by edits is +/// appended to text_tail (number tokens) and arena_tail (strings); floats that +/// are not finite become null, as dump() writes them. +inline void compact_nodes(const document_data& d, std::size_t arena_size, std::vector& out, std::string& text_tail, std::string& arena_tail) +{ + struct frame + { + const node* cur; + const node* end; + std::size_t index; ///< the container's node in out + std::uint32_t count; + bool object; + }; + std::vector stack; + const auto string_node = [&](const node & s) + { + node r = s; + r.extra = 0; + r.flags = static_cast(s.flags & node_flags::storage); + if (r.flags == node_flags::edited) + { + r.off = static_cast(arena_size + arena_tail.size()); + arena_tail.append(d.str(s), s.len); + r.flags = node_flags::escaped; + } + return r; + }; + const auto emit = [&](const node * v) + { + node r = *v; + switch (static_cast(v->kind)) + { + case value_t::object: + case value_t::array: + r.flags = 0; + r.extra = 0; + r.off = (v->flags & (node_flags::moved | node_flags::is_new)) != 0 ? 0 : v->off; + r.len = 0; // counted below + r.next = 0; // set when the container is complete + stack.push_back(frame{d.first_child_edited(v), d.child_end_edited(v), out.size(), 0, v->kind == static_cast(value_t::object)}); + break; + case value_t::string: + r = string_node(*v); + break; + case value_t::number_integer: + case value_t::number_unsigned: + if ((v->flags & node_flags::storage) == node_flags::edited) + { + r.off = static_cast(d.size + text_tail.size()); + text_tail.append(d.str(*v), number_length(*v)); + } + r.flags = 0; + break; + case value_t::number_float: + if ((v->flags & node_flags::storage) == node_flags::edited) + { + const char* const t = d.str(*v); + if (t[0] == 'n' || t[0] == 'i' || (v->len > 1 && t[1] == 'i')) + { + r = node{}; // nan and infinity: null, as dump() writes them + r.kind = static_cast(value_t::null); + break; + } + r.off = static_cast(d.size + text_tail.size()); + text_tail.append(t, v->len); + r.extra = 0xFFFFu; // the digit layout is not recorded + } + r.flags = 0; + break; + case value_t::boolean: + r.flags = static_cast(v->flags & node_flags::is_true); + break; + case value_t::null: + case value_t::binary: + case value_t::discarded: + default: + r.flags = 0; + break; + } + out.push_back(r); + }; + emit(d.tape); + while (!stack.empty()) + { + frame& top = stack.back(); + if (top.cur == top.end) + { + node& c = out[top.index]; + c.len = top.count; + c.next = static_cast(out.size() - top.index); + stack.pop_back(); + continue; + } + ++top.count; + const node* v = nullptr; + if (top.object) + { + out.push_back(string_node(*top.cur)); + v = document_data::deref(top.cur + 1); + top.cur = document_data::after(top.cur + 1); + } + else + { + v = document_data::deref(top.cur); + top.cur = document_data::after(top.cur); + } + emit(v); // may grow the stack (top is not used afterwards) + } +} + +/// the document as an image +inline std::vector save_image(const document_data& d) +{ +#if !NLOHMANN_VIEW_LITTLE_ENDIAN + throw_type_error(320, "json_document images need a little-endian target"); // LCOV_EXCL_LINE +#endif + const std::size_t arena_size = d.arena_size; + const node* nodes = d.tape; + std::size_t count = d.tape_size; + std::vector compacted; + std::string text_tail; + std::string arena_tail; + if (d.edits) + { + compact_nodes(d, arena_size, compacted, text_tail, arena_tail); + nodes = compacted.data(); + count = compacted.size(); + } + const std::size_t text_size = d.size + text_tail.size(); + const std::size_t total_arena = arena_size + arena_tail.size(); + if (NLOHMANN_VIEW_UNLIKELY(text_size >= image_limit || total_arena >= image_limit || count >= image_limit)) + { + // LCOV_EXCL_START (4 GiB) + throw_out_of_range(416, "images of 4 GiB or more are not supported by json_document"); + // LCOV_EXCL_STOP + } + image_header h{}; + h.magic = {{'N', 'J', 'V', 'I'}}; + h.version = image_version; + h.node_count = count; + h.text_size = text_size; + h.arena_size = total_arena; + std::vector image(sizeof(h) + (count * sizeof(node)) + text_size + 1 + total_arena + 1); + std::uint8_t* o = image.data(); + std::memcpy(o, &h, sizeof(h)); + o += sizeof(h); + std::memcpy(o, nodes, count * sizeof(node)); + // the hash indexes are rebuilt by load() + for (std::size_t i = 0; i < count; ++i) + { + if (nodes[i].kind == static_cast(value_t::object) && nodes[i].extra != 0) + { + node n = nodes[i]; + n.extra = 0; + std::memcpy(o + (i * sizeof(node)), &n, sizeof(node)); + } + } + o += count * sizeof(node); + const auto append = [&o](const char* s, std::size_t n) + { + if (n != 0) + { + std::memcpy(o, s, n); + o += n; + } + }; + append(d.src, d.size); + append(text_tail.data(), text_tail.size()); + *o++ = 0; + append(d.base[1], arena_size); + append(arena_tail.data(), arena_tail.size()); + *o = 0; + return image; +} + +/// whether a number node matches its token the way the parser records it +/// (after the bounds check) +inline bool check_number(const node& n, const unsigned char* text) +{ + const std::size_t len = number_length(n); + const unsigned char* const s = text + n.off; + const unsigned char* const e = s + len; + const unsigned char* p = s; + const bool negative = *p == '-'; + p += negative ? 1 : 0; + const unsigned char* const int_start = p; + if (p == e) + { + return false; + } + if (*p == '0') + { + ++p; + } + else if (*p >= '1' && *p <= '9') + { + while (p != e && is_digit(*p)) + { + ++p; + } + } + else + { + return false; + } + const auto int_digits = static_cast(p - int_start); + std::size_t frac_digits = 0; + bool is_float = false; + if (p != e && *p == '.') + { + const unsigned char* const f0 = ++p; + while (p != e && is_digit(*p)) + { + ++p; + } + if (p == f0) + { + return false; + } + frac_digits = static_cast(p - f0); + is_float = true; + } + long exponent = 0; + if (p != e && (*p | 0x20u) == 'e') + { + ++p; + const bool exp_negative = p != e && *p == '-'; + p += (p != e && (*p == '+' || *p == '-')) ? 1 : 0; + if (p == e || !is_digit(*p)) + { + return false; + } + while (p != e && is_digit(*p)) + { + exponent = exponent < 100000 ? (exponent * 10) + (*p - '0') : exponent; + ++p; + } + exponent = exp_negative ? -exponent : exponent; + is_float = true; + } + if (p != e) + { + return false; + } + if (n.kind == static_cast(value_t::number_float)) + { + // the digit layout the parser records (or "many", as compaction + // writes it), and a finite value + const auto layout = static_cast((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8u)); + if (n.extra != layout && n.extra != 0xFFFFu) + { + return false; + } + // parse() rejects floats that overflow; as there, only a number whose + // magnitude could reach 1e308 needs the conversion + if (static_cast(int_digits) + exponent > 300) + { + const auto v = float_value(reinterpret_cast(s), n); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return v <= (std::numeric_limits::max)() && v >= -(std::numeric_limits::max)(); + } + return true; + } + // integers: the token's value is the stored one; number_integer nodes of + // edits can be non-negative (as basic_json keeps the type of a value) + const bool integer = n.kind == static_cast(value_t::number_integer); + if (is_float || int_digits > 20 || (negative && !integer)) + { + return false; + } + // (at most 19 digits cannot overflow; 20 digits are compared with 2^64 - 1) + if (int_digits == 20 && std::memcmp(int_start, "18446744073709551615", 20) > 0) + { + return false; + } + std::uint64_t m = 0; + for (const unsigned char* d = int_start; d != int_start + int_digits; ++d) + { + m = (m * 10) + static_cast(*d - '0'); + } + if (integer && m > (negative ? std::uint64_t{1} << 63u : (std::uint64_t{1} << 63u) - 1)) + { + return false; + } + return integer_bits(n) == (negative ? 0 - m : m); +} + +/// Check the nodes of a loaded image against its text and decoded strings: +/// kinds, flags, and `extra`; extents and element counts of arrays and +/// objects; keys; bounds; string contents (source strings as the parser +/// leaves them: no quotes, backslashes, or control characters; all strings +/// valid UTF-8); and number tokens. +inline bool check_image(const node* nodes, std::size_t count, const unsigned char* text, std::size_t text_size, + const unsigned char* arena, std::size_t arena_size, bool full) +{ + struct frame + { + std::size_t end; + std::uint32_t len; + std::uint32_t seen; + bool object; + bool expect_key; + }; + std::vector stack; + const auto check_string = [&](const node & n) -> bool + { + if ((n.flags & ~node_flags::escaped) != 0 || n.extra != 0) + { + return false; + } + const bool decoded = (n.flags & node_flags::escaped) != 0; + const unsigned char* const base = decoded ? arena : text; + const std::size_t limit = decoded ? arena_size : text_size; + if (n.off > limit || n.len > limit - n.off) + { + return false; + } + if (!full) + { + return true; + } + const unsigned char* const b = base + n.off; + return decoded ? valid_utf8_prefix(b, n.len) == n.len : scan_string_run(b, b + n.len) == b + n.len; + }; + // bounds of a number token; the recorded digit layout must lie within it + const auto number_in_bounds = [&](const node & n) -> bool + { + const std::size_t len = number_length(n); + if (len == 0 || n.off > text_size || len > text_size - n.off) + { + return false; + } + if (n.kind != static_cast(value_t::number_float)) + { + return (n.extra >> 8u) == 0; + } + // float_value() reads the sign, the integer digits, and the point and + // fraction digits the layout records (a layout of more than 19 digits + // means the general conversion, which stays within the token) + const std::size_t int_digits = n.extra & 0xFFu; + const std::size_t frac_digits = n.extra >> 8u; + const std::size_t need = (text[n.off] == '-' ? 1u : 0u) + int_digits + (frac_digits != 0 ? frac_digits + 1 : 0); + return int_digits + frac_digits > 19 || need <= len; + }; + std::size_t i = 0; + for (;;) + { + // close finished arrays and objects + while (!stack.empty() && i == stack.back().end) + { + const frame f = stack.back(); + if (f.seen != f.len || (f.object && !f.expect_key)) + { + return false; + } + stack.pop_back(); + if (!stack.empty()) + { + ++stack.back().seen; + stack.back().expect_key = true; + } + } + if (i == count) + { + return stack.empty(); + } + if (i != 0 && stack.empty()) + { + return false; // nodes after the root + } + const node& n = nodes[i]; + if (!stack.empty() && stack.back().object && stack.back().expect_key) + { + if (n.kind != static_cast(value_t::string) || !check_string(n)) + { + return false; + } + stack.back().expect_key = false; + ++i; + continue; + } + bool complete = true; + switch (static_cast(n.kind)) + { + case value_t::null: + // (the offset of a literal is read to size the output of dump()) + if (n.flags != 0 || n.extra != 0 || n.off > text_size) + { + return false; + } + break; + case value_t::boolean: + if ((n.flags & ~node_flags::is_true) != 0 || n.extra != 0 || n.off > text_size) + { + return false; + } + break; + case value_t::string: + if (!check_string(n)) + { + return false; + } + break; + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + if (n.flags != 0 || !number_in_bounds(n) || (full && !check_number(n, text))) + { + return false; + } + break; + case value_t::array: + case value_t::object: + { + const std::size_t limit = stack.empty() ? count : stack.back().end; + if (n.flags != 0 || n.extra != 0 || n.next == 0 || n.next > limit - i || n.off > text_size) + { + return false; + } + stack.push_back(frame{i + n.next, n.len, 0, n.kind == static_cast(value_t::object), true}); + complete = false; + break; + } + case value_t::binary: + case value_t::discarded: + default: + return false; + } + ++i; + if (complete && !stack.empty()) + { + ++stack.back().seen; + stack.back().expect_key = true; + } + } +} + +[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_invalid_image(const char* what) +{ + throw_parse_error(116, concat("invalid json_document image: ", what)); +} + +/// Read an image into d. The text and the decoded strings stay in the image; +/// the nodes are copied (so that they are aligned, and edits can change them). +inline void load_image(document_data& d, const std::uint8_t* image, std::size_t size, image_check check) +{ +#if !NLOHMANN_VIEW_LITTLE_ENDIAN + throw_type_error(320, "json_document images need a little-endian target"); // LCOV_EXCL_LINE +#endif + if (image == nullptr || size < sizeof(image_header)) + { + throw_invalid_image("too short"); + } + image_header h{}; + std::memcpy(&h, image, sizeof(h)); + // (the reserved fields are for later versions) + if (std::memcmp(h.magic.data(), "NJVI", 4) != 0 || h.version != image_version + || (h.reserved[0] | h.reserved[1] | h.reserved[2] | h.reserved[3]) != 0) + { + throw_invalid_image("unknown format"); + } + const std::size_t room = size - sizeof(h); + if (h.node_count == 0 || h.node_count > room / sizeof(node) || h.node_count >= image_limit || h.text_size >= image_limit || h.arena_size >= image_limit) + { + throw_invalid_image("sizes out of range"); + } + const auto count = static_cast(h.node_count); + const auto text_size = static_cast(h.text_size); + const auto arena_size = static_cast(h.arena_size); + const std::size_t text_at = sizeof(h) + (count * sizeof(node)); + // the text, a NUL, the decoded strings, a NUL, and nothing after them + if (size - text_at < 2 || text_size > size - text_at - 2 || arena_size != size - text_at - text_size - 2 + || image[text_at + text_size] != 0 || image[size - 1] != 0) + { + throw_invalid_image("sizes out of range"); + } + + d.discarded = true; + d.edits.reset(); + d.base[2] = nullptr; + d.owned.clear(); + if (d.owned_image.empty() || image != d.owned_image.data()) + { + d.owned_image.clear(); + } + d.arena.clear(); + d.indexes.clear(); + d.index_slots.clear(); + d.large_objects.clear(); + d.tape_size = 0; + d.reserve(count); + std::memcpy(d.tape, image + sizeof(h), count * sizeof(node)); + d.tape_size = count; + const std::uint8_t* const text = image + text_at; + const std::uint8_t* const arena = text + text_size + 1; + d.src = reinterpret_cast(text); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + d.size = text_size; + d.base[0] = d.src; + d.base[1] = reinterpret_cast(arena); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + d.arena_size = arena_size; + if (check != image_check::none && !check_image(d.tape, count, text, text_size, arena, arena_size, check == image_check::full)) + { + throw_invalid_image("the check failed"); + } + // the hash indexes of large objects, as after parsing + for (std::size_t i = 0; i < count; ++i) + { + node& n = d.tape[i]; + if (n.kind == static_cast(value_t::object)) + { + n.extra = 0; + if (n.len >= document_data::index_min_members) + { + d.large_objects.push_back(static_cast(i)); + } + } + } + build_object_indexes(d); + d.discarded = false; +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -4158,192 +4994,6 @@ NLOHMANN_JSON_NAMESPACE_END // #include // #include -// __ _____ _____ _____ -// __| | __| | | | JSON for Modern C++ -// | | |__ | | | | | | version 3.12.0 -// |_____|_____|_____|_|___| https://github.com/nlohmann/json -// -// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann -// SPDX-License-Identifier: MIT - - - -#include // size_t -#include // int64_t, uint64_t -#include // numeric_limits -#include // string -#include // integral_constant - -// #include -// #include - -// #include - -// #include - -// #include - - -NLOHMANN_JSON_NAMESPACE_BEGIN -namespace detail -{ -namespace view -{ - -/*! -@brief the value of the float token of a node, as parse() converts it - -Uses the lexer's conversion (detail::convert_float), so that the values are -bit-identical to parse(): float and double are converted without allocation -and independent of the locale. The digit layout recorded while parsing locates -the decimal point and the exponent without scanning the token. -*/ -template -NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) -{ - const char* const last = first + n.len; - const std::size_t neg = first[0] == '-' ? 1 : 0; - const std::size_t int_digits = n.extra & 0xFFu; - const std::size_t frac_digits = n.extra >> 8u; - std::size_t dot = std::string::npos; - std::size_t mantissa_end = n.len; - if (int_digits != 255 && frac_digits != 255) - { - dot = frac_digits != 0 ? neg + int_digits : std::string::npos; - mantissa_end = neg + int_digits + (frac_digits != 0 ? 1 + frac_digits : 0); - } - else - { - // more digits than the layout records: locate them - for (std::size_t i = 0; i < n.len; ++i) - { - if (first[i] == '.') - { - dot = i; - } - else if (first[i] == 'e' || first[i] == 'E') - { - mantissa_end = i; - break; - } - } - } - return convert_float(first, last, dot, mantissa_end); -} - -/*! -@brief the digits of a float token with at most 19 digits, from its layout - -The digit layout recorded while parsing says where the integer digits, the -fraction digits, and the exponent are, so the digits are read eight at a -time without scanning. - -@param[in] p first character of the token -@param[in] e end of the token -@param[in] limit end of the readable memory (the source text) -*/ -NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept -{ - const bool negative = *p == '-'; - p += negative ? 1 : 0; - std::uint64_t w = parse_upto19(p, int_digits, limit); - p += int_digits; - std::int64_t q = 0; - if (frac_digits != 0) - { - w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit); - p += 1 + frac_digits; - q = -static_cast(frac_digits); - } - if (p != e) - { - // [eE][+-]digits; huge exponents saturate (the parser rejected overflow) - ++p; - const bool exp_negative = *p == '-'; - p += (*p == '-' || *p == '+') ? 1 : 0; - std::int64_t exp_value = 0; - for (; p != e; ++p) - { - if (exp_value < 0x10000000) - { - exp_value = (exp_value * 10) + (*p - '0'); - } - } - q += exp_negative ? -exp_value : exp_value; - } - - float_significand d; - d.w = w; - d.exponent = q; - d.negative = negative; - return d; -} - -/*! -@brief the value of a float token with at most 19 digits, from its layout - -The result is correctly rounded by the lexer's conversion -(detail::decimal_to_float(): Clinger's fast path where both operands are -exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19 -digits), so it is the value parse() produces. -*/ -template -NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept -{ - return decimal_to_float(layout_decimal(p, e, int_digits, frac_digits, limit)); -} - -/// the value of a float set by an edit: its token (the shortest round-trip -/// text, or "nan", "inf", "-inf") in the edit arena -template -NLOHMANN_VIEW_NOINLINE FloatType edited_float(const char* token, const node& n) -{ - if (token[0] == 'n') - { - return std::numeric_limits::quiet_NaN(); - } - if (token[0] == 'i' || (token[0] == '-' && token[1] == 'i')) - { - return token[0] == 'i' ? std::numeric_limits::infinity() : -std::numeric_limits::infinity(); - } - return float_value(token, n); -} - -/// the value of the float token of a node, as parse() converts it; floats and -/// doubles with at most 19 digits are converted from the digit layout -template -FloatType float_value(const document_data& d, const node& n) -{ - if (NLOHMANN_VIEW_UNLIKELY((n.flags & node_flags::storage) == node_flags::edited)) - { - return edited_float(d.str(n), n); - } - return float_value(d, n, std::integral_constant::value> {}); -} - -template -FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/) -{ - const unsigned int_digits = n.extra & 0xFFu; - const unsigned frac_digits = n.extra >> 8u; - if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many") - { - // (a float token not written by an edit is in the text) - const auto* const first = reinterpret_cast(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) - return layout_float(first, first + n.len, int_digits, frac_digits, reinterpret_cast(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) - } - return float_value(d.str(n), n); -} - -template -FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/) -{ - return float_value(d.str(n), n); -} - -} // namespace view -} // namespace detail -NLOHMANN_JSON_NAMESPACE_END NLOHMANN_JSON_NAMESPACE_BEGIN @@ -4967,7 +5617,9 @@ class view_serializer m_out.put('"'); } - /// as serializer::dump_escaped() for valid UTF-8 (the view has no other) + /// as serializer::dump_escaped(); strings of a document are valid UTF-8, + /// except in a damaged image loaded with image_check::bounds, for which + /// this throws what basic_json::dump() throws for the string template void write_escaped(const unsigned char* s, std::size_t n) { @@ -4991,12 +5643,13 @@ class view_serializer } std::uint32_t codepoint = s[i]; std::size_t len = 1; - if (codepoint >= 0xC0) + if (codepoint >= 0x80) { - len = 2; - if (codepoint >= 0xE0) + len = validate_one_utf8(s + i, n - i); + if (NLOHMANN_VIEW_UNLIKELY(len == 0)) { - len = codepoint >= 0xF0 ? 4 : 3; + invalid_utf8(s, n); + return; } codepoint &= 0xFFu >> (len + 1); for (std::size_t k = 1; k < len; ++k) @@ -5009,6 +5662,13 @@ class view_serializer } } + /// throw what basic_json::dump() throws for a string that is not valid UTF-8 + NLOHMANN_VIEW_NOINLINE static void invalid_utf8(const unsigned char* s, std::size_t n) + { + const string_t dumped = BasicJsonType(string_t(reinterpret_cast(s), n)).dump(); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + static_cast(dumped); + } + template void write_codepoint(std::uint32_t codepoint, const unsigned char* bytes, std::size_t len) { @@ -6168,7 +6828,7 @@ class basic_json_document /// whether the document holds its own copy of the text bool owns_source() const noexcept { - return m_data && !m_data->owned.empty() && m_data->src == m_data->owned.data(); + return m_data && ((!m_data->owned.empty() && m_data->src == m_data->owned.data()) || !m_data->owned_image.empty()); } /// number of index nodes (values plus object keys) @@ -6177,7 +6837,7 @@ class basic_json_document return m_data ? m_data->tape_size : 0; } - /// bytes held by the document (index, decoded strings, owned text) + /// bytes held by the document (index, decoded strings, owned text or image) std::size_t memory_usage() const noexcept { if (!m_data) @@ -6186,7 +6846,7 @@ class basic_json_document } return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node)) + (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0) - + m_data->arena.capacity() + m_data->owned.capacity() + + m_data->arena.capacity() + m_data->owned.capacity() + m_data->owned_image.capacity() + (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t)) + (m_data->large_objects.capacity() * sizeof(std::uint32_t)) + (m_data->edits != nullptr ? m_data->edits->bytes : 0); @@ -6206,8 +6866,10 @@ class basic_json_document // allocate everything first, so that an exception leaves the document // unchanged + // (the decoded strings of a loaded image stay in the image) + const bool arena_in_use = d.base[1] == d.arena.data(); const bool shrink_arena = d.arena.capacity() > d.arena.size(); - std::string arena(shrink_arena ? d.arena : std::string()); + std::string arena(shrink_arena && arena_in_use ? d.arena : std::string()); // (edits link to the nodes of the index, which then stays in place) const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap && d.edits == nullptr; const bool into_header = d.tape_size <= d.inline_cap; @@ -6223,10 +6885,62 @@ class basic_json_document if (shrink_arena) { d.arena.swap(arena); - d.base[1] = d.arena.data(); + if (arena_in_use) + { + d.base[1] = d.arena.data(); + } } } + //////////// + // images // + //////////// + + /// how load() checks an image (full, bounds, or none) + using image_check = detail::view::image_check; + + /// The document as an image that load() reads without parsing: the node + /// index, the text, and the decoded strings. An edited document is + /// written in its current state (floats that are not finite become null, + /// as in dump()). + std::vector save() const + { + if (NLOHMANN_VIEW_UNLIKELY(!m_data || m_data->discarded)) + { + detail::view::throw_type_error(320, "cannot save a discarded json_document"); + } + return detail::view::save_image(*m_data); + } + + /// Read an image written by save(). The image is borrowed: it must stay + /// alive and unchanged while the document is used. + NLOHMANN_VIEW_NODISCARD + static basic_json_document load(const std::uint8_t* image, std::size_t size, const image_check check = image_check::full) + { + basic_json_document d; + d.ensure_data(nullptr, 0); + detail::view::load_image(*d.m_data, image, size, check); + return d; + } + + /// read an image (borrowed) + NLOHMANN_VIEW_NODISCARD + static basic_json_document load(const std::vector& image, const image_check check = image_check::full) + { + return load(image.data(), image.size(), check); + } + + /// read an image and keep it (no copy) + NLOHMANN_VIEW_NODISCARD + static basic_json_document load(std::vector&& image, const image_check check = image_check::full) + { + basic_json_document d; + d.ensure_data(nullptr, 0); + d.m_data->owned_image = std::move(image); + detail::view::load_image(*d.m_data, d.m_data->owned_image.data(), d.m_data->owned_image.size(), check); + return d; + } + /////////// // edits // /////////// @@ -6410,6 +7124,7 @@ class basic_json_document { d.owned.clear(); } + d.owned_image.clear(); d.src = src; d.size = size; d.tape_size = 0; @@ -6434,6 +7149,7 @@ class basic_json_document { d.base[0] = d.src; d.base[1] = d.arena.data(); + d.arena_size = d.arena.size(); detail::view::build_object_indexes(d); d.discarded = false; return; diff --git a/tests/Makefile b/tests/Makefile index e8e61e603..e03b9dc7d 100644 --- a/tests/Makefile +++ b/tests/Makefile @@ -10,7 +10,7 @@ CXXFLAGS += -std=c++11 CPPFLAGS += -I ../single_include FUZZER_ENGINE = src/fuzzer-driver_afl.cpp -FUZZERS = parse_afl_fuzzer parse_bson_fuzzer parse_cbor_fuzzer parse_msgpack_fuzzer parse_ubjson_fuzzer parse_bjdata_fuzzer parse_bon8_fuzzer parse_json_view_fuzzer +FUZZERS = parse_afl_fuzzer parse_bson_fuzzer parse_cbor_fuzzer parse_msgpack_fuzzer parse_ubjson_fuzzer parse_bjdata_fuzzer parse_bon8_fuzzer parse_json_view_fuzzer json_view_image_fuzzer fuzzers: $(FUZZERS) parse_afl_fuzzer: @@ -19,6 +19,9 @@ parse_afl_fuzzer: parse_json_view_fuzzer: $(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-parse_json_view.cpp -o $@ +json_view_image_fuzzer: + $(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-json_view_image.cpp -o $@ + parse_bson_fuzzer: $(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-parse_bson.cpp -o $@ diff --git a/tests/fuzzing.md b/tests/fuzzing.md index 8ef983229..69433b742 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -10,6 +10,12 @@ produces, and that a rejected input makes both parsers throw with an identical ` reuses the `corpus_json` corpus (or, for the `make fuzz_testing_json_view` target below, `tests/data/json_tests`) rather than a format of its own. +`json_view_image_fuzzer` (`tests/src/fuzzer-json_view_image.cpp`) tests the images of `json_document` (`save()` and +`load()`). It uses each input twice: as an image, which `load()` must either reject with `parse_error.116` or read +safely (with `image_check::full`, the document must also serialize to the JSON it reads as), and as a JSON text, whose +image must load and serialize to the same text. A corpus of images can be made from JSON files with a small program +that calls `json_document::parse(text).save()`; plain JSON files work as well. + ## Corpus creation For most effective fuzzing, a [corpus](https://llvm.org/docs/LibFuzzer.html#corpus) should be provided. A corpus is a diff --git a/tests/src/fuzzer-json_view_image.cpp b/tests/src/fuzzer-json_view_image.cpp new file mode 100644 index 000000000..39c340d2e --- /dev/null +++ b/tests/src/fuzzer-json_view_image.cpp @@ -0,0 +1,89 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +/* +This file implements a test of json_document images suitable for fuzz +testing. The input is used twice: + +- as an image: json_document::load() with image_check::full must either throw + a parse_error or yield a document that serializes to the JSON text it reads + as; with image_check::bounds, reading and serializing must be safe (checked + by the sanitizers), and serializing may only throw type_error.316 +- as a JSON text: if json_document::parse() accepts it, the image of the + document must load (with every check) and serialize to the same text + +The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer +drivers. +*/ + +#include +#include +#include +#include +#include +#include + +// the checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + +using json = nlohmann::json; +using json_document = nlohmann::json_document; +using image_check = json_document::image_check; + +// see http://llvm.org/docs/LibFuzzer.html +extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +{ + // the input as an image + for (const image_check check : {image_check::full, image_check::bounds}) + { + json_document d; + try + { + d = json_document::load(data, size, check); + } + catch (const json::parse_error& e) + { + assert(e.id == 116); + continue; + } + std::string dumped; + try + { + dumped = d.root().dump(); + } + catch (const json::type_error& e) + { + // invalid UTF-8 can only pass the bounds check + assert(check == image_check::bounds && e.id == 316); + continue; + } + const json j = d.root().materialize(); + if (check == image_check::full) + { + assert(json::parse(dumped) == j); + // an image of the loaded document is the input + assert(d.save() == std::vector(data, data + size)); + } + } + + // the input as a JSON text + const std::string text(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const json_document parsed = json_document::parse(text, false); + if (!parsed.is_discarded()) + { + const std::vector image = parsed.save(); + for (const image_check check : {image_check::full, image_check::bounds, image_check::none}) + { + const json_document loaded = json_document::load(image, check); + assert(loaded.root().dump() == parsed.root().dump()); + } + } + return 0; +} diff --git a/tests/src/unit-json_view_image.cpp b/tests/src/unit-json_view_image.cpp new file mode 100644 index 000000000..b2fb05a4b --- /dev/null +++ b/tests/src/unit-json_view_image.cpp @@ -0,0 +1,791 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +using nlohmann::json; +using nlohmann::ordered_json; +using nlohmann::json_document; +using nlohmann::json_editable_document; +using nlohmann::ordered_json_document; +using nlohmann::ordered_json_editable_document; +using image_check = json_document::image_check; +using nlohmann::detail::view::node; + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#if !(defined(__BYTE_ORDER__) && defined(__ORDER_BIG_ENDIAN__) && __BYTE_ORDER__ == __ORDER_BIG_ENDIAN__) + +namespace +{ +std::string exception_of(const std::function& f) +{ + try + { + f(); + } + catch (const json::exception& e) + { + return e.what(); + } + return ""; +} + +const char* const check_failed = "[json.exception.parse_error.116] parse error: invalid json_document image: the check failed"; + +std::string read_file(const std::string& name) +{ + std::ifstream f(std::string(TEST_DATA_DIRECTORY) + name, std::ios::binary); + std::stringstream ss; + ss << f.rdbuf(); + return ss.str(); +} + +// the offsets of the parts of an image +constexpr std::size_t header_size = 64; + +std::uint64_t header_field(const std::vector& image, std::size_t offset) +{ + std::uint64_t v = 0; + std::memcpy(&v, image.data() + offset, sizeof(v)); + return v; +} + +void set_header_field(std::vector& image, std::size_t offset, std::uint64_t v) +{ + std::memcpy(image.data() + offset, &v, sizeof(v)); +} + +std::size_t node_count(const std::vector& image) +{ + return static_cast(header_field(image, 8)); +} + +std::size_t text_at(const std::vector& image) +{ + return header_size + (node_count(image) * sizeof(node)); +} + +node node_at(const std::vector& image, std::size_t i) +{ + node n{}; + std::memcpy(&n, image.data() + header_size + (i * sizeof(node)), sizeof(node)); + return n; +} + +void set_node(std::vector& image, std::size_t i, const node& n) +{ + std::memcpy(image.data() + header_size + (i * sizeof(node)), &n, sizeof(node)); +} + +/// the result of loading an image with a check: "" or the exception message +std::string load_result(const std::vector& image, image_check check) +{ + return exception_of([&] + { + const json_document d = json_document::load(image, check); + static_cast(d); + }); +} + +/// a copy of the image with node i changed by f +template +std::vector corrupted(const std::vector& image, std::size_t i, F f) +{ + std::vector b = image; + node n = node_at(b, i); + f(n); + set_node(b, i, n); + return b; +} + +/// a document and the documents loaded from its image must be equal +template +void check_round_trip(const Document& d) +{ + const std::vector image = d.save(); + for (const image_check check : + { + image_check::full, image_check::bounds, image_check::none + }) + { + const json_document l = json_document::load(image, check); + CHECK(l.root().dump() == d.root().dump()); + CHECK(l.root().dump(2) == d.root().dump(2)); + CHECK(l.root().materialize() == json(d.root().materialize())); + // an image of a loaded document is the same image + CHECK(l.save() == image); + } + // an editable document can be loaded, too + const ordered_json_editable_document e = ordered_json_editable_document::load(image); + CHECK(e.root().dump() == d.root().dump()); +} + +std::uint32_t rng() +{ + static std::mt19937 generator(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed): reproducible + return generator(); +} +} // namespace + +TEST_CASE("json_view images: round trips") +{ + SECTION("small documents") + { + for (const char* text : + { + "null", "true", "false", "0", "-0", "42", "-42", "18446744073709551615", "-9223372036854775808", + "123456789012345678901234567890", "1.5", "-1.25e-300", "1E308", "0.1000000000000000000000000001", + "\"\"", "\"text\"", R"("esc\"aped\n\u00e9\ud83d\ude00")", "\"\xc3\xa9\xe3\x81\x82\"", + "[]", "{}", "[[]]", "[{}]", "{\"\":{}}", + R"({"a": [1, 2.5, "x\ty", true, null, {"b": []}], "c": {"d": -3, "eA": "f"}})", + R"({"k": 1, "k": 2, "l": [], "k": 3})", + " [1 , 2 ] " + }) + { + CAPTURE(text); + check_round_trip(json_document::parse(text)); + check_round_trip(ordered_json_document::parse(text)); + } + } + + SECTION("files") + { + for (const char* name : + { + "/json_testsuite/sample.json", "/nativejson-benchmark/canada.json", "/nativejson-benchmark/citm_catalog.json", + "/nativejson-benchmark/twitter.json", "/json_tests/pass1.json", "/json_tests/pass2.json", "/json_tests/pass3.json" + }) + { + CAPTURE(name); + const std::string text = read_file(name); + const json_document d = json_document::parse(text); + check_round_trip(d); + // what a loaded document reads is what parse() produces + CHECK(json_document::load(d.save()).root().materialize() == json::parse(text)); + } + } + + SECTION("images are deterministic") + { + const std::string text = R"({"b": [1, 2, {"c": "\u00e9"}], "a": 1.5})"; + const json_document d = json_document::parse(text); + CHECK(d.save() == json_document::parse(text).save()); + CHECK(d.save() == json_editable_document::parse(text).save()); + CHECK(d.save() == ordered_json_document::parse(text).save()); + const json_document copy = json_document::parse_copy(text); + CHECK(copy.save() == d.save()); + } + + SECTION("large objects get their hash index again") + { + std::string text = "{"; + for (int i = 0; i < 1000; ++i) + { + text += (i != 0 ? ",\"k" : "\"k") + std::to_string(i) + "\":" + std::to_string(i); + } + text += R"(,"k7":"a duplicate","inner":{)"; + for (int i = 0; i < 200; ++i) + { + text += (i != 0 ? ",\"m" : "\"m") + std::to_string(i) + "\":" + std::to_string(-i); + } + text += "}}"; + const json_document d = json_document::parse(text); + const std::vector image = d.save(); + for (const image_check check : + { + image_check::full, image_check::none + }) + { + const json_document l = json_document::load(image, check); + for (int i = 0; i < 1000; ++i) + { + CHECK(l.root()["k" + std::to_string(i)] == d.root()["k" + std::to_string(i)]); + } + CHECK(l.root()["k7"].get() == 7); // the first of duplicate keys + CHECK(l.root()["inner"]["m199"].get() == -199); + CHECK(!l.root().contains("k1000")); + // the index is not part of the image + CHECK(l.save() == image); + } + // the nodes of objects in the image do not carry the number of an index + CHECK(node_at(image, 0).extra == 0); + } +} + +TEST_CASE("json_view images: edited documents") +{ + const std::string text = R"({"name": "x", "n": 1, "f": 2.5, "list": [1, 2, 3], "obj": {"a": "\u00e9", "b": [true]}, "s": "a\"b"})"; + + SECTION("every kind of edit") + { + ordered_json_editable_document d = ordered_json_editable_document::parse(text); + d.set(d.root()["name"], "a new \"name\""); // string in the edit arena + d.set(d.root()["n"], -17); // negative integer + d.set(d.root(), "p", 5); // non-negative number_integer + d.set(d.root(), "u", 18446744073709551615u); // unsigned + d.set(d.root()["f"], 0.1); // float token + d.set(d.root(), "nan", std::numeric_limits::quiet_NaN()); + d.set(d.root(), "inf", -std::numeric_limits::infinity()); + d.push_back(d.root()["list"], "pushed"); // moved array + d.insert(d.root()["list"], 0, ordered_json::object({{"new", {1, 2}}})); + d.erase(d.root()["list"], 2); + d.erase(d.root(), "s"); + d.set(d.root()["obj"], "c", ordered_json::array({1, "two", 3.5, nullptr, false})); // new object member with a new array + d.set(d.root(), "copy", d.root()["obj"]); // a copy of a subtree + d.set(d.root(), "key \xc3\xa9", true); // a key in the edit arena + + const std::vector image = d.save(); + const ordered_json expected = ordered_json::parse(d.root().dump()); + for (const image_check check : + { + image_check::full, image_check::bounds, image_check::none + }) + { + const ordered_json_document l = ordered_json_document::load(image, check); + CHECK(l.root().dump() == d.root().dump()); + CHECK(l.root().materialize() == expected); + CHECK(l.root()["nan"].is_null()); + CHECK(l.root()["inf"].is_null()); + CHECK(l.root()["p"].is_number_integer()); + CHECK(l.root()["p"].get() == 5); + CHECK(l.root()["f"].get() == 0.1); + CHECK(l.root()["u"].get() == 18446744073709551615u); + } + // the node index is in document order again: an image of the loaded + // document is the same image + CHECK(ordered_json_document::load(image).save() == image); + // number tokens of edits follow the source; the text is the source's + // prefix + const ordered_json_document l = ordered_json_document::load(image); + REQUIRE(l.source().size() > text.size()); + CHECK(std::string(l.source().data(), text.size()) == text); + } + + SECTION("a loaded document can be edited and saved again") + { + const std::vector first = json_editable_document::parse(text).save(); + json_editable_document d = json_editable_document::load(first); + d.set(d.root()["obj"]["a"], "changed"); + d.push_back(d.root()["list"], 4); + d.set(d.root(), "z", json::array({json::object()})); + const std::vector second = d.save(); + const json_document l = json_document::load(second); + CHECK(l.root().dump() == d.root().dump()); + CHECK(l.root()["obj"]["a"] == "changed"); + CHECK(l.root()["list"].size() == 4); + } + + SECTION("the root replaced") + { + json_editable_document d = json_editable_document::parse(text); + d.set(d.root(), json::array({1, "x"})); + check_round_trip(d); + d.set(d.root(), 3.5); + check_round_trip(d); + d.set(d.root(), "text"); + check_round_trip(d); + } +} + +TEST_CASE("json_view images: ownership") +{ + const std::string text = R"({"a": "esc\u00e9aped", "b": [1, 2]})"; + const std::vector image = json_document::parse(text).save(); + + SECTION("borrowed") + { + const json_document d = json_document::load(image); + CHECK(!d.owns_source()); + CHECK(d.root()["a"] == "esc\xc3\xa9" "aped"); + const json_document p = json_document::load(image.data(), image.size()); + CHECK(!p.owns_source()); + CHECK(p.root() == d.root()); + // the text is the image's + CHECK(d.source().data() == reinterpret_cast(image.data() + text_at(image))); + } + + SECTION("owned") + { + std::vector copy = image; + const std::uint8_t* const data = copy.data(); + json_document d = json_document::load(std::move(copy)); + CHECK(d.owns_source()); + CHECK(d.source().data() == reinterpret_cast(data + text_at(image))); + CHECK(d.memory_usage() >= image.size()); + CHECK(d.root()["b"][1] == 2); + // read() replaces the image + d.read(std::string("[1]")); + CHECK(d.owns_source()); + CHECK(d.root().dump() == "[1]"); + const std::string borrowed = "[2]"; + d.read(borrowed); + CHECK(!d.owns_source()); + } + + SECTION("shrink_to_fit keeps the decoded strings of the image") + { + json_document d = json_document::parse(R"(["\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9\u00e9"])"); + d.shrink_to_fit(); + const std::vector img = d.save(); + json_document l = json_document::load(img); + l.shrink_to_fit(); + CHECK(l.root().dump() == d.root().dump()); + CHECK(l.save() == img); + } +} + +TEST_CASE("json_view images: errors") +{ + SECTION("a literal as the root: dump() after loading") + { + for (const char* text : + { + "null", "true", "false" + }) + { + std::vector image = json_document::parse(text).save(); + node n = node_at(image, 0); + n.off = static_cast(image.size()); + set_node(image, 0, n); + CHECK(load_result(image, image_check::full) == check_failed); + CHECK(load_result(image, image_check::bounds) == check_failed); + } + } + + SECTION("saving a discarded document") + { + const json_document empty; + CHECK(exception_of([&] { static_cast(empty.save()); }) == "[json.exception.type_error.320] cannot save a discarded json_document"); + const json_document failed = json_document::parse("[1,", false); + CHECK(exception_of([&] { static_cast(failed.save()); }) == "[json.exception.type_error.320] cannot save a discarded json_document"); + } + + const std::vector image = json_document::parse(R"({"a": [1, "\u00e9"]})").save(); + const std::string prefix = "[json.exception.parse_error.116] parse error: invalid json_document image: "; + + SECTION("header and sizes") + { + CHECK(exception_of([] + { + const json_document d = json_document::load(nullptr, 0); + static_cast(d); + }) == prefix + "too short"); + CHECK(exception_of([&] + { + const json_document d = json_document::load(image.data(), 63); + static_cast(d); + }) == prefix + "too short"); + + std::vector bad = image; + bad[0] = 'X'; + CHECK(load_result(bad, image_check::full) == prefix + "unknown format"); + bad = image; + bad[4] = 2; // version + CHECK(load_result(bad, image_check::full) == prefix + "unknown format"); + for (std::size_t reserved = 32; reserved < 64; reserved += 8) + { + bad = image; + bad[reserved + 3] = 1; + CHECK(load_result(bad, image_check::none) == prefix + "unknown format"); + } + + const auto sizes = [&](std::size_t offset, std::uint64_t v) + { + std::vector b = image; + set_header_field(b, offset, v); + return load_result(b, image_check::none); + }; + CHECK(sizes(8, 0) == prefix + "sizes out of range"); // no nodes + CHECK(sizes(8, 1000) == prefix + "sizes out of range"); // more nodes than bytes + CHECK(sizes(8, 0xFFFFFFF0u) == prefix + "sizes out of range"); + CHECK(sizes(16, 0xFFFFFFF0u) == prefix + "sizes out of range"); // text size + CHECK(sizes(16, header_field(image, 16) + 1) == prefix + "sizes out of range"); + CHECK(sizes(16, image.size()) == prefix + "sizes out of range"); + CHECK(sizes(24, 0xFFFFFFF0u) == prefix + "sizes out of range"); // decoded string size + CHECK(sizes(24, header_field(image, 24) - 1) == prefix + "sizes out of range"); + + // the NULs after the text and the decoded strings + bad = image; + bad[text_at(image) + header_field(image, 16)] = 'x'; + CHECK(load_result(bad, image_check::none) == prefix + "sizes out of range"); + bad = image; + bad.back() = 'x'; + CHECK(load_result(bad, image_check::none) == prefix + "sizes out of range"); + // nothing after the image + bad = image; + bad.push_back(0); + CHECK(load_result(bad, image_check::none) == prefix + "sizes out of range"); + // nodes, but not even room for the NULs + bad.assign(image.begin(), image.begin() + static_cast(text_at(image))); + CHECK(load_result(bad, image_check::none) == prefix + "sizes out of range"); + } +} + +TEST_CASE("json_view images: check") +{ + // nodes: 0 { 1 "s" 2 "x\"y" (escaped) 3 "i" 4 -12 5 "u" 6 7 7 "f" 8 1.5e300 9 "b" 10 true 11 "n" 12 null + // 13 "a" 14 [ 15 "t" 16 {} ] + const std::string text = R"({"s":"x\"y","i":-12,"u":7,"f":1.5e300,"b":true,"n":null,"a":["t",{}]})"; + const std::vector image = json_document::parse(text).save(); + REQUIRE(load_result(image, image_check::full).empty()); + REQUIRE(node_count(image) == 17); + + // bounds: rejected by both checks; content: only by the full one + const auto rejected = [&](const std::vector& b, bool bounds) + { + CHECK(load_result(b, image_check::full) == check_failed); + CHECK(load_result(b, image_check::bounds) == (bounds ? check_failed : "")); + }; + + SECTION("kinds") + { + const std::array kinds = {{8, 9, 10, 200}}; // binary, discarded, link, unknown + for (const std::uint8_t kind : kinds) + { + rejected(corrupted(image, 12, [&](node & n) + { + n.kind = kind; + }), true); + } + // a key that is not a string + rejected(corrupted(image, 1, [](node & n) + { + n.kind = 0; + n.len = 0; + n.off = 0; + }), true); + } + + SECTION("flags and extra") + { + rejected(corrupted(image, 12, [](node & n) + { + n.flags = 4; + }), true); + rejected(corrupted(image, 12, [](node & n) + { + n.extra = 1; + }), true); + rejected(corrupted(image, 10, [](node & n) + { + n.flags = 5; + }), true); + rejected(corrupted(image, 10, [](node & n) + { + n.extra = 1; + }), true); + rejected(corrupted(image, 1, [](node & n) + { + n.flags = 2; // a string in the edit arena + }), true); + rejected(corrupted(image, 1, [](node & n) + { + n.extra = 3; + }), true); + rejected(corrupted(image, 4, [](node & n) + { + n.flags = 2; + }), true); + rejected(corrupted(image, 4, [](node & n) + { + n.extra = static_cast(n.extra | 0x100u); // an integer with fraction digits + }), true); + rejected(corrupted(image, 0, [](node & n) + { + n.flags = 8; // moved + }), true); + rejected(corrupted(image, 0, [](node & n) + { + n.extra = 1; // a hash index + }), true); + } + + SECTION("bounds") + { + const std::size_t text_size = header_field(image, 16); + const std::size_t arena_size = header_field(image, 24); + rejected(corrupted(image, 1, [&](node & n) + { + n.off = static_cast(text_size + 1); + }), true); + rejected(corrupted(image, 1, [&](node & n) + { + n.len = static_cast(text_size); + }), true); + rejected(corrupted(image, 2, [&](node & n) + { + n.len = static_cast(arena_size + 1); + }), true); + rejected(corrupted(image, 6, [&](node & n) + { + n.off = static_cast(text_size); + }), true); + rejected(corrupted(image, 6, [&](node & n) + { + n.off = static_cast(text_size + 5); + }), true); + rejected(corrupted(image, 6, [](node & n) + { + n.extra = 0; // no digits + }), true); + rejected(corrupted(image, 8, [&](node & n) + { + n.len = static_cast(text_size); + }), true); + rejected(corrupted(image, 8, [](node & n) + { + n.len = 2; // shorter than the recorded digits + }), true); + rejected(corrupted(image, 14, [&](node & n) + { + n.off = static_cast(text_size + 1); + }), true); + // literals: their offset sizes the output of dump() + rejected(corrupted(image, 10, [&](node & n) + { + n.off = static_cast(text_size + 1); + }), true); + rejected(corrupted(image, 12, [&](node & n) + { + n.off = 0xFFFFFFFFu; + }), true); + } + + SECTION("structure") + { + rejected(corrupted(image, 0, [](node & n) + { + n.next = 0; + }), true); + rejected(corrupted(image, 0, [](node & n) + { + n.next = 18; // beyond the image + }), true); + rejected(corrupted(image, 14, [](node & n) + { + n.next = 4; // beyond the enclosing object + }), true); + rejected(corrupted(image, 0, [](node & n) + { + n.len = 6; // member count + }), true); + rejected(corrupted(image, 14, [](node & n) + { + n.len = 3; // element count + }), true); + rejected(corrupted(image, 0, [](node & n) + { + n.next = 14; // the object ends after the key "a" + n.len = 7; + }), true); + rejected(corrupted(image, 0, [](node & n) + { + n.next = 13; // nodes after the root + n.len = 6; + }), true); + rejected(corrupted(image, 0, [](node & n) + { + n.kind = 2; // an array: the "keys" are values, and the counts do not match + }), true); + const std::vector as_array = corrupted(image, 16, [](node & n) + { + n.kind = 2; // {} as []: fine + }); + CHECK(load_result(as_array, image_check::full).empty()); + CHECK(json_document::load(as_array).root().dump() == R"({"s":"x\"y","i":-12,"u":7,"f":1.5e+300,"b":true,"n":null,"a":["t",[]]})"); + } + + SECTION("strings") + { + // a quote in a source string (the full check only) + std::vector b = image; + const std::size_t t = text_at(image); + const node t15 = node_at(image, 15); + b[t + t15.off] = '"'; + rejected(b, false); + // a control character + b[t + t15.off] = '\n'; + rejected(b, false); + // invalid UTF-8 in a decoded string + b = image; + const node s2 = node_at(image, 2); + b[t + header_field(image, 16) + 1 + s2.off] = 0xFF; + rejected(b, false); + } + + SECTION("numbers") + { + const std::size_t t = text_at(image); + const node i4 = node_at(image, 4); + const node u6 = node_at(image, 6); + const node f8 = node_at(image, 8); + const auto at_token = [&](const node & n, std::size_t k, std::uint8_t c) + { + std::vector b = image; + b[t + n.off + k] = c; + return b; + }; + rejected(at_token(i4, 1, 'x'), false); // -x2 + rejected(at_token(i4, 1, '0'), false); // -02 + rejected(at_token(i4, 0, '1'), false); // 112 != -12 + rejected(at_token(f8, 1, 'x'), false); // 1x5e300 + rejected(at_token(f8, 2, 'e'), false); // 1.ee300 + rejected(at_token(f8, 4, 'x'), false); // 1.5ex00 + rejected(at_token(f8, 3, '0'), false); // 1.50300: another layout + rejected(at_token(f8, 4, '9'), false); // 1.5e900: overflow + rejected(at_token(f8, 0, 'x'), false); + rejected(at_token(u6, 0, '8'), false); // 8 != 7 + rejected(corrupted(image, 4, [](node & n) + { + n.kind = 6; // "-12" as unsigned: a sign + n.extra = 3; + }), false); + // a non-negative number_integer (as edits write it): fine + const std::vector positive = corrupted(image, 6, [](node & n) + { + n.kind = 5; + n.extra = 0; + }); + CHECK(load_result(positive, image_check::full).empty()); + CHECK(json_document::load(positive).root()["u"].is_number_integer()); + rejected(corrupted(image, 8, [](node & n) + { + n.kind = 6; // a float token as integer + n.extra = 7; + }), false); + } + + SECTION("float tokens of an image checked for bounds only") + { + // A float node whose layout records "many" digits is converted from + // its token alone; a token that is not a JSON number reads as 0. + const std::vector img = json_document::parse("[1.5e300,2]").save(); + const std::size_t t = text_at(img); + for (const char* token : + { + "x.5e300", "01.5e30", "1.xe300", "1.5ex00", "1.5e+x0", "1.5e30x", "-.5e300", "1.5E300" + }) + { + CAPTURE(token); + std::vector b = img; + node n = node_at(b, 1); + n.extra = 0xFFFFu; + set_node(b, 1, n); + std::memcpy(b.data() + t + n.off, token, n.len); + const json_document d = json_document::load(b, image_check::bounds); + const auto v = d.root()[0].get(); + CHECK(v == (std::string(token) == "1.5E300" ? 1.5e300 : 0.0)); + CHECK(load_result(b, image_check::full) == (std::string(token) == "1.5E300" ? "" : check_failed)); + } + } + + SECTION("integer ranges") + { + // tokens of many digits, which the parser stores as floats + const std::string big = R"([123456789012345678901234, 99999999999999999999, 9223372036854775808])"; + const std::vector img = json_document::parse(big).save(); + const auto as_integer = [&](std::size_t i, std::uint8_t kind, std::uint16_t extra) + { + std::vector b = img; + node n = node_at(b, i); + n.kind = kind; + n.extra = extra; + set_node(b, i, n); + return load_result(b, image_check::full); + }; + CHECK(as_integer(1, 6, 24) == check_failed); // more than 20 digits + CHECK(as_integer(2, 6, 20) == check_failed); // more than 2^64 - 1 + CHECK(as_integer(3, 5, 18) == check_failed); // more than 2^63 - 1 as number_integer + } +} + +TEST_CASE("json_view images: damaged images") +{ + // A damaged image must be rejected, or read safely; with the full check, + // it also serializes to the JSON it reads as. + const std::vector texts = + { + R"({"a": [1, -2, 3.25, "x\u00e9y", true, null], "b": {"c": "\"q\"", "d": 1e10}, "e": ""})", + R"([[[[]]], {"k": {"k": {"k": 12345678901234567890}}}, "\ud83d\ude00", -0.0, 0])", + }; + for (const std::string& text : texts) + { + const std::vector image = json_document::parse(text).save(); + for (int round = 0; round < 3000; ++round) + { + std::vector b = image; + const std::uint32_t flips = 1 + (rng() % 3); + for (std::uint32_t k = 0; k < flips; ++k) + { + // mostly the nodes, where the damage matters most + const std::size_t at = rng() % 4 != 0 ? header_size + (rng() % (b.size() - header_size)) : rng() % b.size(); + b[at] = static_cast(rng() % 3 == 0 ? rng() : b[at] ^ (1u << (rng() % 8))); + } + for (const image_check check : + { + image_check::full, image_check::bounds + }) + { + json_document d; + try + { + d = json_document::load(b, check); + } + catch (const json::parse_error& e) + { + CHECK(e.id == 116); + continue; + } + std::string dumped; + std::string dumped_ascii; + try + { + dumped = d.root().dump(); + dumped_ascii = d.root().dump(-1, ' ', true); + } + catch (const json::type_error& e) + { + // invalid UTF-8 (the bounds check only) + CHECK(check == image_check::bounds); + CHECK(e.id == 316); + continue; + } + const json j = d.root().materialize(); + if (check == image_check::full) + { + CHECK(json::parse(dumped) == j); + CHECK(json::parse(dumped_ascii) == j); + } + } + } + } +} + +#else + +TEST_CASE("json_view images: big-endian targets") +{ + const json_document d = json_document::parse("[1]"); + CHECK_THROWS_WITH_AS(d.save(), "[json.exception.type_error.320] json_document images need a little-endian target", json::type_error&); +} + +#endif