diff --git a/BUILD.bazel b/BUILD.bazel index 945ac488e..51d6525be 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -66,6 +66,7 @@ cc_library( "include/nlohmann/detail/string_escape.hpp", "include/nlohmann/detail/string_utils.hpp", "include/nlohmann/detail/value_t.hpp", + "include/nlohmann/detail/view/builder.hpp", "include/nlohmann/detail/view/document_data.hpp", "include/nlohmann/detail/view/macro_scope.hpp", "include/nlohmann/detail/view/macro_unscope.hpp", diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp new file mode 100644 index 000000000..9756afbc6 --- /dev/null +++ b/include/nlohmann/detail/view/builder.hpp @@ -0,0 +1,1026 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2020 YaoYuan +// SPDX-License-Identifier: MIT + +#pragma once + +#include // find, find_if, max +#include // size_t, ptrdiff_t +#include // uint8_t, uint16_t, uint32_t, uint64_t +#include // memcmp, memcpy +#include // numeric_limits +#include // string +#include // vector + +#include +#include +#include +#include +#include + +// The view's parser: one pass over the input that emits the node index (see +// node.hpp). The table-driven decoding of \u escapes and the fast paths for +// ": " and indentation follow yyjson (https://github.com/ibireme/yyjson, MIT +// license). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +enum class error_code : std::uint8_t +{ + none, + empty_input, + unexpected_value, + invalid_literal, + expected_key, + expected_colon, + expected_array_end, + expected_object_end, + trailing_characters, + number_after_minus, + number_after_dot, + number_after_exponent, + number_overflow, + string_missing_quote, + string_control_character, + string_utf8, + string_escape, + string_unicode_hex, + string_surrogate_high, + string_surrogate_low, + comment_start, + comment_unterminated, + input_too_large, +}; + +struct parse_failure +{ + error_code code = error_code::none; + std::size_t offset = 0; ///< byte offset of the offending character +}; + +/// FloatType: the number_float_t of the document, whose overflow parse() rejects +template +class builder +{ + public: + builder(document_data& d, const char* src, std::size_t size) noexcept + : doc(d) + , b(reinterpret_cast(src)) + , e(b + size) + {} + + /// returns false and fills `failure` on error + bool run() + { + cursor c(*this); + return c.run(); + } + + builder(const builder&) = delete; + builder& operator=(const builder&) = delete; + + parse_failure failure{}; + + private: + struct frame + { + std::uint32_t idx; + std::uint32_t count; + bool is_object; + }; + + document_data& doc; + const unsigned char* const b; + const unsigned char* const e; + + // the open array/object is in the cursor; enclosing ones on a stack that is + // inline for the first 64 levels + frame shallow[64]; + std::vector deep{}; + + NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept + { + failure.code = c; + failure.offset = static_cast(at - b); + doc.tape_size = 0; + return false; + } + + /// a failure (recorded by fail) as a comment() result + const unsigned char* fail_at(error_code c, const unsigned char* at) noexcept + { + fail(c, at); + return nullptr; + } + + /// a decoded string: p after its closing quote (nullptr: an error), and + /// its bytes in the arena + struct decoded + { + const unsigned char* p; + std::size_t start; + std::size_t len; + }; + + decoded failed(error_code c, const unsigned char* at) noexcept + { + fail(c, at); + return decoded{nullptr, 0, 0}; + } + + /// the comment at p (*p == '/'): the position after it, or nullptr on error + NLOHMANN_VIEW_NOINLINE const unsigned char* comment(const unsigned char* p) + { + if (e - p < 2) + { + ++p; + return fail_at(error_code::comment_start, p); + } + if (p[1] == '/') + { + p += 2; + while (p != e && *p != '\n' && *p != '\r' && !(NulIsEnd && *p == 0)) + { + ++p; + } + if (NulIsEnd && p != e && *p == 0) + { + ++p; // as in parse(), a null byte ends the comment like a line break + } + return p; + } + if (p[1] == '*') + { + p += 2; + for (;;) + { + if (p == e || (NulIsEnd && *p == 0)) + { + return fail_at(error_code::comment_unterminated, p); + } + if (*p == '*' && p + 1 != e && p[1] == '/') + { + p += 2; + return p; + } + ++p; + } + } + ++p; + return fail_at(error_code::comment_start, p); + } + + /// the index is full (n nodes, parsed up to at): extrapolate the node + /// count from the nodes per input byte so far (with headroom, and at least + /// 1.5 times as many), so that dense inputs regrow once instead of + /// doubling repeatedly; returns the new node array + NLOHMANN_VIEW_NOINLINE node* grow(std::size_t n, const unsigned char* at) + { + const std::uint64_t done = static_cast(at - b) + 1; + const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; + doc.tape_size = n; + doc.reserve((std::max)(static_cast(guess + guess / 4 + 64), n + n / 2 + 64)); + return doc.tape; + } + + /// four hex digits at p as a code unit (p moves past them), or -1 (p at + /// the first bad digit); one table lookup per digit and a single check, + /// as in yyjson's read_hex_u16 + NLOHMANN_VIEW_ALWAYS_INLINE int hex4(const unsigned char*& p) noexcept + { + // digit values; 0xF0: not a hex digit + static const std::uint8_t hex[256] = + { + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 10, 11, 12, 13, 14, 15, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 10, 11, 12, 13, 14, 15, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + }; + if (NLOHMANN_VIEW_LIKELY(e - p >= 4)) + { + const unsigned d0 = hex[p[0]], d1 = hex[p[1]], d2 = hex[p[2]], d3 = hex[p[3]]; + if (NLOHMANN_VIEW_LIKELY(((d0 | d1 | d2 | d3) & 0xF0u) == 0)) + { + p += 4; + return static_cast((d0 << 12) | (d1 << 8) | (d2 << 4) | d3); + } + } + p = hex4_error(p); + return -1; + } + + /// hex4() failed: the first bad digit (none: p) + NLOHMANN_VIEW_NOINLINE const unsigned char* hex4_error(const unsigned char* p) noexcept + { + if (e - p >= 4) + { + while (is_hex(*p)) + { + ++p; + } + } + return p; + } + + /// escapes present (or an error) in the string at s, scanned up to p: + /// decode into the arena + NLOHMANN_VIEW_NOINLINE decoded slow_string(const unsigned char* s, const unsigned char* p) + { + // single-character escapes; 0: invalid (and 'u', handled separately) + static const char simple_escape[128] = + { + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, '"', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '/', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '\\', 0, 0, 0, + 0, 0, '\b', 0, 0, 0, '\f', 0, 0, 0, 0, 0, 0, 0, '\n', 0, 0, 0, '\r', 0, '\t', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + }; + const std::size_t start = arena_used(); + arena_run(s, static_cast(p - s)); + for (;;) + { + if (p == e) + { + return failed(error_code::string_missing_quote, p); + } + const unsigned char c = *p; + if (c == '"') + { + ++p; + return decoded{p, start, arena_used() - start}; + } + if (c != '\\') + { + if (NulIsEnd && c == 0) + { + return failed(error_code::string_missing_quote, p); + } + return failed(c < 0x20 ? error_code::string_control_character : error_code::string_utf8, p); + } + ++p; + if (p == e) + { + return failed(error_code::string_missing_quote, p); + } + const unsigned char d = *p++; + arena_ensure(4); + if (d == 'u') + { + int cp = hex4(p); + if (NLOHMANN_VIEW_UNLIKELY(cp < 0)) + { + return failed(error_code::string_unicode_hex, p); + } + if (NLOHMANN_VIEW_UNLIKELY((cp & 0xF800) == 0xD800)) // a surrogate + { + if (cp >= 0xDC00) + { + return failed(error_code::string_surrogate_low, p); + } + if (e - p < 2 || p[0] != '\\' || p[1] != 'u') + { + return failed(error_code::string_surrogate_high, p); + } + p += 2; + const int lo = hex4(p); + if (lo < 0) + { + return failed(error_code::string_unicode_hex, p); + } + if (lo < 0xDC00 || lo > 0xDFFF) + { + return failed(error_code::string_surrogate_high, p); + } + cp = 0x10000 + ((cp - 0xD800) << 10) + (lo - 0xDC00); + } + aw = put_utf8(aw, cp); + } + else if (NLOHMANN_VIEW_LIKELY(d < 128 && simple_escape[d] != 0)) + { + *aw++ = simple_escape[d]; + } + else + { + --p; + return failed(error_code::string_escape, p); + } + if (p != e && *p == '\\') + { + continue; // consecutive escapes ("\u00e4\u00f6"): no run in between + } + const unsigned char* const r = p; + p = scan_string_run(p, e); + arena_run(r, static_cast(p - r)); + } + } + + /// does the float token [s, p) overflow FloatType? (parse() rejects it) + NLOHMANN_VIEW_NOINLINE static bool float_overflows(const unsigned char* s, const unsigned char* p) + { + const auto* const first = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const auto* const last = reinterpret_cast(p); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const char* const dot = std::find(first, last, '.'); + const char* const exponent = std::find_if(first, last, [](char c) + { + return c == 'e' || c == 'E'; + }); + const auto v = convert_float(first, last, dot == last ? std::string::npos : static_cast(dot - first), + static_cast(exponent - first)); + return v > (std::numeric_limits::max)() || v < -(std::numeric_limits::max)(); + } + + /// does the magnitude digits [d, d + n) exceed the given limit (same length)? + static bool digits_exceed(const unsigned char* d, const char* limit, std::size_t n) noexcept + { + return std::memcmp(d, limit, n) > 0; + } + + static bool is_hex(unsigned char c) noexcept + { + return (c >= '0' && c <= '9') || (c >= 'A' && c <= 'F') || (c >= 'a' && c <= 'f'); + } + + // decode arena: the std::string in the document, written through a raw + // pointer (resized ahead in large steps; trimmed when parsing succeeds) + char* aw = nullptr; + char* aend = nullptr; + + NLOHMANN_VIEW_ALWAYS_INLINE std::size_t arena_used() const noexcept + { + return aw != nullptr ? static_cast(aw - doc.arena.data()) : 0; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void arena_ensure(std::size_t n) + { + if (NLOHMANN_VIEW_UNLIKELY(static_cast(aend - aw) < n)) + { + arena_grow(n); + } + } + + NLOHMANN_VIEW_NOINLINE void arena_grow(std::size_t n) + { + const std::size_t used = arena_used(); + doc.arena.resize((std::max)(doc.arena.size() * 2, used + n + 256)); + aw = &doc.arena[0] + used; + aend = &doc.arena[0] + doc.arena.size(); + } + + /// append the run [r, r + n) to the arena; short runs as one fixed-size + /// 16-byte move when both sides have the room (no library call) + NLOHMANN_VIEW_ALWAYS_INLINE void arena_run(const unsigned char* r, std::size_t n) + { + arena_ensure(n + 16); + if (n <= 16 && e - r >= 16) + { + std::memcpy(aw, r, 16); + } + else + { + std::memcpy(aw, r, n); + } + aw += n; + } + + /// UTF-8 encoding of cp at w (room for 4 bytes) + static char* put_utf8(char* w, int cp) noexcept + { + if (cp < 0x80) + { + *w++ = static_cast(cp); + } + else if (cp < 0x800) + { + *w++ = static_cast(0xC0 | (cp >> 6)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + else if (cp < 0x10000) + { + *w++ = static_cast(0xE0 | (cp >> 12)); + *w++ = static_cast(0x80 | ((cp >> 6) & 0x3F)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + else + { + *w++ = static_cast(0xF0 | (cp >> 18)); + *w++ = static_cast(0x80 | ((cp >> 12) & 0x3F)); + *w++ = static_cast(0x80 | ((cp >> 6) & 0x3F)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + return w; + } + + /// The parse state and the parser proper. The cursor is a local object of + /// run() whose address never escapes (everything it calls out of line is a + /// member of the builder and gets the positions it needs), so that the + /// compiler keeps the state in registers instead of reloading it from + /// memory after every node store and call. + struct cursor + { + explicit cursor(builder& owner) noexcept + : cold(owner) + , b(owner.b) + , p(owner.b) + , e(owner.e) + {} + + builder& cold; ///< out-of-line helpers and state that needs no registers + const unsigned char* const b; + const unsigned char* p; + const unsigned char* const e; + node* base = nullptr; + node* out = nullptr; + node* cap = nullptr; + + // the open array/object + std::uint32_t cur_idx = 0; + std::uint32_t cur_count = 0; + bool cur_is_object = false; + std::size_t depth = 0; + + NLOHMANN_VIEW_ALWAYS_INLINE bool run() + { + cold.doc.reserve(estimate_nodes(reinterpret_cast(b), static_cast(e - b))); + base = cold.doc.tape; + out = base; + cap = base + cold.doc.tape_cap; + + if (e - p >= 3 && p[0] == 0xEF && p[1] == 0xBB && p[2] == 0xBF) + { + p += 3; // byte order mark + } + if (!ws()) + { + return false; + } + if (p == e || (NulIsEnd && *p == 0)) + { + return fail(error_code::empty_input); + } + + // root value + switch (cur()) + { + case '{': + open(value_t::object); + ++p; + goto obj_first; + case '[': + open(value_t::array); + ++p; + goto arr_first; + default: + if (!scalar()) + { + return false; + } + goto root_done; + } + + // value dispatch, expanded once for array elements and once for member + // values: each jump has its own history (arrays tend to hold one kind of + // value), and the continuation needs no branch on the container kind +#define NLOHMANN_VIEW_VALUE(NEXT) \ + switch (cur()) \ + { \ + case '"': \ + if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \ + goto NEXT; \ + case '{': \ + open(value_t::object); \ + ++p; \ + goto obj_first; \ + case '[': \ + open(value_t::array); \ + ++p; \ + goto arr_first; \ + case '-': \ + if (NLOHMANN_VIEW_UNLIKELY(!number())) { return false; } \ + goto NEXT; \ + case '0': case '1': case '2': case '3': \ + case '4': case '5': case '6': case '7': case '8': case '9': \ + if (NLOHMANN_VIEW_UNLIKELY(!number())) { return false; } \ + goto NEXT; \ + case 't': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal("true", 4, value_t::boolean, node_flags::is_true))) { return false; } \ + goto NEXT; \ + case 'f': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal_false())) { return false; } \ + goto NEXT; \ + case 'n': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal("null", 4, value_t::null, 0))) { return false; } \ + goto NEXT; \ + default: \ + return fail(error_code::unexpected_value); \ + } + +arr_first: + if (!ws()) + { + return false; + } + if (cur() == ']') + { + ++p; + goto close_container; + } +value: + NLOHMANN_VIEW_VALUE(arr_next) +arr_next: + ++cur_count; + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ',')) + { + ++p; + if (!ws()) + { + return false; + } + if (TrailingCommas && cur() == ']') + { + ++p; + goto close_container; + } + goto value; + } + if (cur() == ']') + { + ++p; + goto close_container; + } + return fail(error_code::expected_array_end); + +obj_first: + if (!ws()) + { + return false; + } + if (cur() == '}') + { + ++p; + goto close_container; + } +obj_key: + if (NLOHMANN_VIEW_UNLIKELY(cur() != '"')) + { + return fail(error_code::expected_key); + } + if (NLOHMANN_VIEW_UNLIKELY(!string())) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ':' && (Sentinel || e - p >= 2) && p[1] == ' ')) + { + p += 2; // ": " (pretty-printed input; a fast path of yyjson) + } + else + { + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_UNLIKELY(cur() != ':')) + { + return fail(error_code::expected_colon); + } + ++p; + } + if (!ws()) + { + return false; + } + NLOHMANN_VIEW_VALUE(obj_next) +obj_next: + ++cur_count; + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ',')) + { + ++p; + if (!ws()) + { + return false; + } + if (TrailingCommas && cur() == '}') + { + ++p; + goto close_container; + } + goto obj_key; + } + if (cur() == '}') + { + ++p; + goto close_container; + } + return fail(error_code::expected_object_end); + +#undef NLOHMANN_VIEW_VALUE + +close_container: + close(); + if (NLOHMANN_VIEW_UNLIKELY(depth == 0)) + { + goto root_done; + } + if (cur_is_object) + { + goto obj_next; + } + goto arr_next; + +root_done: + if (!ws()) + { + return false; + } + if (p != e && !(NulIsEnd && *p == 0)) + { + return fail(error_code::trailing_characters); + } + cold.doc.tape_size = static_cast(out - base); + cold.doc.arena.resize(cold.arena_used()); + return true; + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool fail(error_code c) noexcept + { + return cold.fail(c, p); + } + + /// the current byte, or 0 at the end. With a NUL-terminated input + /// (Sentinel) the terminator is read instead of checking the bounds; a 0 + /// never matches a JSON token, so the error paths tell the end apart. + NLOHMANN_VIEW_ALWAYS_INLINE unsigned char cur() const noexcept + { + return Sentinel ? *p : (p != e ? *p : 0); + } + + /// root scalar + NLOHMANN_VIEW_ALWAYS_INLINE bool scalar() + { + switch (cur()) + { + case '"': + return string(); + case 't': + return literal("true", 4, value_t::boolean, node_flags::is_true); + case 'f': + return literal_false(); + case 'n': + return literal("null", 4, value_t::null, 0); + case '-': + return number(); + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + return number(); + default: + return fail(error_code::unexpected_value); + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool literal_false() + { + return literal("false", 5, value_t::boolean, 0); + } + + /// skip whitespace (and comments); false on a malformed comment + NLOHMANN_VIEW_ALWAYS_INLINE bool ws() + { + const unsigned char c = cur(); + if (NLOHMANN_VIEW_LIKELY(c > ' ' && (!Comments || c != '/'))) + { + return true; // no whitespace: the common case in minified input + } + return ws_slow(); + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool ws_slow() + { + for (;;) + { + if (cur() == ' ' && (Sentinel || e - p >= 2) && p[1] > ' ' && (!Comments || p[1] != '/')) + { + ++p; // single space, e.g. after ':' or ',' + return true; + } + if (cur() == '\n' || cur() == '\r') + { + // (a branch, not an add of the comparison: p must not + // wait for the byte after the line break) + if (NLOHMANN_VIEW_LIKELY(cur() == '\n')) + { + ++p; + } + else if ((Sentinel || e - p >= 2) && p[1] == '\n') + { + p += 2; + } + else + { + ++p; + } + // indentation: two spaces per step, fixed offsets (after yyjson) + while (e - p >= 32) + { +#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(load16(p + 2 * (i)) == 0x2020)) {} else { p += 2 * (i); goto indent_done; } + NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) +#undef NLOHMANN_VIEW_STEP + p += 32; + } +indent_done: + ; + } + for (unsigned char c = cur(); c == ' ' || c == '\n' || c == '\r' || c == '\t'; c = cur()) + { + ++p; + } + if (Comments && cur() == '/') + { + const unsigned char* const q = cold.comment(p); + if (q == nullptr) + { + return false; + } + p = q; + continue; + } + return true; + } + } + + /// append a node: (kind, flags, extra, off) and the second word (len, or + /// an integer's value); two stores on little-endian targets + NLOHMANN_VIEW_ALWAYS_INLINE node* emit(value_t k, std::uint8_t flags, std::uint16_t extra, std::size_t off, std::uint64_t second) + { + if (NLOHMANN_VIEW_UNLIKELY(out == cap)) + { + const std::size_t n = static_cast(out - base); + base = cold.grow(n, p); + out = base + n; + cap = base + cold.doc.tape_cap; + } + node* n = out++; +#if NLOHMANN_VIEW_LITTLE_ENDIAN + const std::uint64_t first = static_cast(k) | (static_cast(flags) << 8) + | (static_cast(extra) << 16) | (static_cast(off) << 32); + std::memcpy(reinterpret_cast(n), &first, 8); + std::memcpy(reinterpret_cast(n) + 8, &second, 8); +#else + n->kind = static_cast(k); + n->flags = flags; + n->extra = extra; + n->off = static_cast(off); + set_integer_bits(*n, second); +#endif + return n; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void open(value_t k) + { + const auto idx = static_cast(emit(k, 0, 0, static_cast(p - b), 0) - base); + if (depth != 0) + { + const frame f = {cur_idx, cur_count, cur_is_object}; + if (NLOHMANN_VIEW_LIKELY(depth <= 64)) + { + cold.shallow[depth - 1] = f; + } + else + { + cold.deep.push_back(f); + } + } + ++depth; + cur_idx = idx; + cur_count = 0; + cur_is_object = k == value_t::object; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void close() + { + node& n = base[cur_idx]; + n.len = cur_count; + n.next = static_cast(out - base) - cur_idx; + if (--depth != 0) + { + frame f; + if (NLOHMANN_VIEW_LIKELY(depth <= 64)) + { + f = cold.shallow[depth - 1]; + } + else + { + f = cold.deep.back(); + cold.deep.pop_back(); + } + cur_idx = f.idx; + cur_count = f.count; + cur_is_object = f.is_object; + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool literal(const char* text, std::size_t n, value_t k, std::uint8_t flags) + { + if (NLOHMANN_VIEW_UNLIKELY(e - p < static_cast(n) || std::memcmp(p, text, n) != 0)) + { + return fail(error_code::invalid_literal); + } + emit(k, flags, 0, static_cast(p - b), n); + p += n; + return true; + } + + /// a number at p; the sign is known from the dispatch (so that p does + /// not have to wait for the first byte) + template + NLOHMANN_VIEW_ALWAYS_INLINE bool number() + { + const unsigned char* const s = p; + if (negative) + { + ++p; + } + const unsigned char* const int_start = p; + if (p != e && *p == '0') + { + ++p; + } + else if (NLOHMANN_VIEW_LIKELY(p != e && *p >= '1' && *p <= '9')) + { + p = skip_digits(p + 1, e); + } + else + { + return fail(error_code::number_after_minus); + } + const std::size_t int_digits = static_cast(p - int_start); + std::size_t frac_digits = 0; + bool is_float = false; + if (p != e && *p == '.') + { + ++p; + const unsigned char* const f0 = p; + p = skip_digits(p, e); + if (NLOHMANN_VIEW_UNLIKELY(p == f0)) + { + return fail(error_code::number_after_dot); + } + frac_digits = static_cast(p - f0); + is_float = true; + } + long exponent = 0; + if (p != e && (*p | 0x20) == 'e') + { + ++p; + bool exp_negative = false; + if (p != e && (*p == '+' || *p == '-')) + { + exp_negative = *p == '-'; + ++p; + } + if (NLOHMANN_VIEW_UNLIKELY(p == e || !is_digit(*p))) + { + return fail(error_code::number_after_exponent); + } + while (p != e && is_digit(*p)) + { + if (exponent < 100000) + { + exponent = exponent * 10 + (*p - '0'); + } + ++p; + } + if (exp_negative) + { + exponent = -exponent; + } + is_float = true; + } + + value_t kind = is_float ? value_t::number_float : (negative ? value_t::number_integer : value_t::number_unsigned); + if (!is_float) + { + // integers that do not fit become floats, as in parse() + if (NLOHMANN_VIEW_UNLIKELY(int_digits >= 19)) + { + if (negative) + { + if (int_digits > 19 || (int_digits == 19 && digits_exceed(int_start, "9223372036854775808", 19))) + { + kind = value_t::number_float; + } + } + else if (int_digits > 20 || (int_digits == 20 && digits_exceed(int_start, "18446744073709551615", 20))) + { + kind = value_t::number_float; + } + } + } + // parse() rejects floats that overflow; only numbers whose magnitude + // could reach the largest FloatType (1e308 for double, 1e38 for + // float) need the conversion + if (NLOHMANN_VIEW_UNLIKELY(static_cast(int_digits) + exponent > std::numeric_limits::max_exponent10 - 8 && kind == value_t::number_float)) + { + if (cold.float_overflows(s, p)) + { + p = s; + return fail(error_code::number_overflow); + } + } + const auto layout = static_cast((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8)); + std::uint64_t second = static_cast(p - s); + if (kind != value_t::number_float) + { + // integers are converted now, while their digits are in cache + const std::uint64_t m = int_digits <= 19 ? parse_upto19(int_start, static_cast(int_digits), e) + : parse_upto19(int_start, 19, e) * 10 + static_cast(int_start[19] - '0'); + second = negative ? 0 - m : m; + } + emit(kind, 0, layout, static_cast(s - b), second); + return true; + } + + /// a string (value or key) at p + NLOHMANN_VIEW_ALWAYS_INLINE bool string() + { + ++p; // opening quote + const unsigned char* const s = p; + p = scan_string_run(p, e); + if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"')) + { + emit(value_t::string, 0, 0, static_cast(s - b), static_cast(p - s)); + ++p; + return true; + } + const decoded r = cold.slow_string(s, p); + if (r.p == nullptr) + { + return false; + } + p = r.p; + emit(value_t::string, node_flags::escaped, 0, r.start, r.len); + return true; + } + }; +}; + +/// run the builder with compile-time options +template +inline bool build_with(document_data& d, const char* src, std::size_t size, bool sentinel, parse_failure& failure) +{ + if (sentinel) + { + builder bld(d, src, size); + const bool ok = bld.run(); + failure = bld.failure; + return ok; + } + builder bld(d, src, size); + const bool ok = bld.run(); + failure = bld.failure; + return ok; +} + +/// sentinel: src[size] is readable and 0 (e.g. std::string); FloatType: the +/// number_float_t of the document +template +inline bool build(document_data& d, const char* src, std::size_t size, bool comments, bool trailing_commas, bool sentinel, parse_failure& failure) +{ + if (comments) + { + return trailing_commas ? build_with(d, src, size, sentinel, failure) + : build_with(d, src, size, sentinel, failure); + } + return trailing_commas ? build_with(d, src, size, sentinel, failure) + : build_with(d, src, size, sentinel, failure); +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp new file mode 100644 index 000000000..9d4223327 --- /dev/null +++ b/tests/src/unit-json_view_builder.cpp @@ -0,0 +1,327 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +#include +using nlohmann::json; + +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace +{ +using nlohmann::detail::view::document_data; +using nlohmann::detail::view::node; + +// the node index of a text; a vector input has no terminating NUL, so that +// AddressSanitizer catches any read past the last byte +struct built +{ + std::unique_ptr data{}; // NOLINT(readability-redundant-member-init) + std::vector copy{}; // NOLINT(readability-redundant-member-init) + bool ok = false; + nlohmann::detail::view::parse_failure failure{}; +}; + +template +built build(const std::string& text, bool comments, bool trailing_commas, bool sentinel) +{ + built r; + r.data.reset(document_data::create(nlohmann::detail::view::estimate_nodes(text.data(), text.size()))); + const char* src = text.c_str(); + if (!sentinel) + { + r.copy.assign(text.begin(), text.end()); + src = r.copy.data(); + } + r.ok = nlohmann::detail::view::build < FloatType, !nlohmann::detail::abi_config::strict_nul_handling > (*r.data, src, text.size(), comments, trailing_commas, sentinel, r.failure); + r.data->src = src; + r.data->base[0] = src; + r.data->base[1] = r.data->arena.data(); + return r; +} + +// the value of a subtree, as json::parse would build it +json value_of(const document_data& d, const node*& n) +{ + const node& x = *n; + ++n; + switch (static_cast(x.kind)) + { + case json::value_t::object: + { + json o = json::object(); + const node* const end = &x + x.next; + while (n != end) + { + const std::string key(d.str(*n), n->len); + ++n; + o[key] = value_of(d, n); + } + return o; + } + case json::value_t::array: + { + json a = json::array(); + const node* const end = &x + x.next; + while (n != end) + { + a.push_back(value_of(d, n)); + } + return a; + } + case json::value_t::string: + return std::string(d.str(x), x.len); + case json::value_t::boolean: + return (x.flags & nlohmann::detail::view::node_flags::is_true) != 0; + case json::value_t::number_integer: + return static_cast(nlohmann::detail::view::integer_bits(x)); + case json::value_t::number_unsigned: + return nlohmann::detail::view::integer_bits(x); + case json::value_t::number_float: + return json::parse(std::string(d.src + x.off, x.len)).get(); + case json::value_t::null: + case json::value_t::binary: + case json::value_t::discarded: + default: + return nullptr; + } +} + +json value_of(const built& b) +{ + const node* n = b.data->tape; + json v = value_of(*b.data, n); + CHECK(n == b.data->tape + b.data->tape_size); + return v; +} + +// accept/reject and the value must match json::parse, for all options and +// with and without a NUL after the text +void check_same(const std::string& text) +{ + CAPTURE(text); + for (int options = 0; options < 4; ++options) + { + const bool comments = (options & 1) != 0; + const bool trailing_commas = (options & 2) != 0; + const bool accepted = json::accept(text, comments, trailing_commas); + for (const bool sentinel : + { + true, false + }) + { + const built b = build(text, comments, trailing_commas, sentinel); + CHECK(b.ok == accepted); + if (b.ok && accepted) + { + CHECK(value_of(b) == json::parse(text, nullptr, true, comments, trailing_commas)); + } + } + } +} + +// a small deterministic generator of documents +struct generator +{ + std::mt19937 rng{5295}; + + int r(int n) + { + return static_cast(rng() % static_cast(n)); + } + + void ws(std::string& o) + { + for (int n = r(4) == 0 ? r(12) : r(2); n > 0; --n) + { + o += " \n\t\r "[r(6)]; + } + } + + void str(std::string& o) + { + static const char* const pieces[] = {"a", "Z", " ", "~", "\\n", "\\\"", "\\\\", "\\/", "\\u00e9", "\\ud83d\\ude00", "\xc3\xa9", "\xe3\x81\x82", "\xf0\x9f\x98\x80", "\x7f", "\\u001f", "long enough text to leave the first 16 bytes"}; + o += '"'; + for (int n = r(3) == 0 ? r(20) : r(6); n > 0; --n) + { + o += pieces[r(16)]; + } + o += '"'; + } + + void num(std::string& o) + { + static const char* const numbers[] = {"0", "-0", "1", "-1", "12", "123456789", "1234567890123456789", "9223372036854775807", "-9223372036854775808", + "9223372036854775808", "18446744073709551615", "18446744073709551616", "-9223372036854775809", + "1.5", "-2.25e-3", "1e10", "1E+2", "0.000001", "3.141592653589793238462643", "1e308", "-1e-400", "123.456e7" + }; + o += numbers[r(22)]; + } + + void value(std::string& o, int depth) + { + ws(o); + const int k = depth > 5 ? 2 + r(6) : r(8); + if (k == 0 || k == 1) + { + const bool object = k == 0; + o += object ? '{' : '['; + for (int i = r(5); i > 0; --i) + { + ws(o); + if (object) + { + str(o); + ws(o); + o += ':'; + } + value(o, depth + 1); + o += i > 1 ? "," : ""; + } + ws(o); + o += object ? '}' : ']'; + } + else if (k < 4) + { + str(o); + } + else if (k < 6) + { + num(o); + } + else + { + static const char* const literals[] = {"true", "false", "null"}; + o += literals[r(3)]; + } + ws(o); + } +}; +} // namespace + +TEST_CASE("json_view builder") +{ + SECTION("scalars and containers") + { + for (const char* text : + { + "null", "true", "false", "0", "-0", "42", "-42", "1.5", "\"\"", "\"abc\"", "[]", "{}", "[1,2,3]", "{\"a\":1,\"b\":[true,null]}", + " [ 1 , 2 ] ", "{\"a\" : {\"b\" : {}}}", "[[[]]]", "\"\\u00e4\\n\\ud83d\\ude00\"", "{\"a\":1,\"a\":2}", "18446744073709551616", + "-9223372036854775809", "123456789012345678901234567890", "1e400", "-1e400", "1.7976931348623157e308" + }) + { + check_same(text); + } + } + + SECTION("malformed input") + { + for (const char* text : + { + "", " ", "[", "]", "{", "}", "[1,]", "{\"a\":1,}", "[1 2]", "{\"a\" 1}", "{1:2}", "tru", "nul", "fals", "truex", "-", "01", "1.", ".5", "1e", "1e+", + "\"", "\"abc", "\"\\x\"", "\"\\u12\"", "\"\\u12G4\"", "\"\\ud800\"", "\"\\udc00\"", "\"\\ud800\\u0041\"", "\"\x01\"", "\"\xff\"", "\"\xc3\"", + "\"\xe0\x80\x80\"", "\"\xed\xa0\x80\"", "[1]x", "[1] [2]", "/", "/*", "/* */ 1", "// c\n1", "1 // c", "[1,/*c*/2]", "[1,2,]" + }) + { + check_same(text); + } + } + + SECTION("NUL, BOM, and whitespace") + { + check_same(std::string("[1]\0garbage", 11)); + check_same(std::string("[1\0]", 4)); + check_same(std::string("[1, // c\0\n2]", 12)); + check_same(std::string("[1, /* c\0 */ 2]", 15)); + check_same("\xEF\xBB\xBF[1]"); + check_same("\xEF\xBB[1]"); + check_same(" \t\r\n 7 \n"); + for (const char* text : + {"[1]\r", "[1]\n", "[1]\r\n", "[1,\r2]", "[1,\r\n2]", "7\r", "\"x\"\r", "{\"a\":\r\n1}\r", "[\n 1,\n 2\n]", "{\n \"a\": [\n 1\n ]\n}" + }) + { + check_same(text); + } + } + + SECTION("deep nesting") + { + // the open containers beyond 64 levels live on the heap + for (const std::size_t depth : + { + 63u, 64u, 65u, 1000u, 100000u + }) + { + const std::string arrays = std::string(depth, '[') + std::string(depth, ']'); + const built b = build(arrays, false, false, false); + REQUIRE(b.ok); + CHECK(b.data->tape_size == depth); + CHECK(b.data->tape[0].next == depth); + std::string objects; + for (std::size_t i = 0; i < depth; ++i) + { + objects += "{\"a\":"; + } + objects += '1' + std::string(depth, '}'); + const built o = build(objects, false, false, false); + REQUIRE(o.ok); + CHECK(o.data->tape_size == (2 * depth) + 1); + CHECK(!build(std::string(depth, '[') + std::string(depth - 1, ']'), false, false, false).ok); + } + } + + SECTION("generated documents and damaged copies") + { + generator g; + for (int i = 0; i < 3000; ++i) + { + std::string text; + g.value(text, 0); + check_same(text); + // damage: flip one byte, or cut the text + std::string damaged = text; + const auto at = static_cast(g.r(static_cast(damaged.size()))); + static const char replacements[] = {'x', '"', '\\', ',', ':', ']', '}', '[', '{', '1', '-', '.', 'e', '\0', '\n', '/'}; + damaged[at] = replacements[g.r(16)]; + check_same(damaged); + check_same(text.substr(0, at)); + } + } + + SECTION("test files") + { + for (const char* name : + { + "/json.org/1.json", "/json.org/2.json", "/json.org/3.json", "/json.org/4.json", "/json.org/5.json", + "/json_testsuite/sample.json", "/nativejson-benchmark/canada.json", "/nativejson-benchmark/citm_catalog.json", + "/nativejson-benchmark/twitter.json", "/json_tests/pass1.json", "/json_tests/pass2.json", "/json_tests/pass3.json" + }) + { + CAPTURE(name); + std::ifstream f(std::string(TEST_DATA_DIRECTORY) + name, std::ios::binary); + std::stringstream ss; + ss << f.rdbuf(); + const std::string text = ss.str(); + REQUIRE(!text.empty()); + const built b = build(text, false, false, true); + REQUIRE(b.ok); + CHECK(value_of(b) == json::parse(text)); + } + } +}