From 23e517a64a9205efd09ece60ebd71023f4892719 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Mon, 28 Sep 2026 21:27:30 +0200 Subject: [PATCH] Add the one-pass parser of json_view (internal) The builder parses JSON text in one pass into the node index: strings and numbers stay in the source (escaped strings are decoded into an arena), integers are converted while their digits are in cache, and floats keep their digit layout for a later conversion. It accepts exactly what json::parse accepts, for every combination of comments and trailing commas, with and without a terminating NUL, and with JSON_STRICT_NUL_HANDLING. Parse state lives in a local cursor whose address never escapes, so that it stays in registers; out-of-line helpers (errors, regrowth, escapes, comments) are members of the builder and get the positions they need. The value dispatch is expanded once for array elements and once for member values. Literals are compared with memcmp and words read in a fixed byte order, so nothing depends on the platform's byte order. Error messages come with the public classes. Tests (unit-json_view_builder.cpp): accept/reject and values against json::parse for handwritten, generated, and damaged documents under all option combinations, from std::string and from exact-size buffers (no read past the input under AddressSanitizer), deep nesting up to 100,000 levels, NUL/BOM/whitespace cases, and the test-suite files. Signed-off-by: Niels Lohmann --- BUILD.bazel | 1 + include/nlohmann/detail/view/builder.hpp | 1026 ++++++++++++++++++++++ tests/src/unit-json_view_builder.cpp | 327 +++++++ 3 files changed, 1354 insertions(+) create mode 100644 include/nlohmann/detail/view/builder.hpp create mode 100644 tests/src/unit-json_view_builder.cpp diff --git a/BUILD.bazel b/BUILD.bazel index 945ac488e..51d6525be 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -66,6 +66,7 @@ cc_library( "include/nlohmann/detail/string_escape.hpp", "include/nlohmann/detail/string_utils.hpp", "include/nlohmann/detail/value_t.hpp", + "include/nlohmann/detail/view/builder.hpp", "include/nlohmann/detail/view/document_data.hpp", "include/nlohmann/detail/view/macro_scope.hpp", "include/nlohmann/detail/view/macro_unscope.hpp", diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp new file mode 100644 index 000000000..9756afbc6 --- /dev/null +++ b/include/nlohmann/detail/view/builder.hpp @@ -0,0 +1,1026 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2020 YaoYuan +// SPDX-License-Identifier: MIT + +#pragma once + +#include // find, find_if, max +#include // size_t, ptrdiff_t +#include // uint8_t, uint16_t, uint32_t, uint64_t +#include // memcmp, memcpy +#include // numeric_limits +#include // string +#include // vector + +#include +#include +#include +#include +#include + +// The view's parser: one pass over the input that emits the node index (see +// node.hpp). The table-driven decoding of \u escapes and the fast paths for +// ": " and indentation follow yyjson (https://github.com/ibireme/yyjson, MIT +// license). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +enum class error_code : std::uint8_t +{ + none, + empty_input, + unexpected_value, + invalid_literal, + expected_key, + expected_colon, + expected_array_end, + expected_object_end, + trailing_characters, + number_after_minus, + number_after_dot, + number_after_exponent, + number_overflow, + string_missing_quote, + string_control_character, + string_utf8, + string_escape, + string_unicode_hex, + string_surrogate_high, + string_surrogate_low, + comment_start, + comment_unterminated, + input_too_large, +}; + +struct parse_failure +{ + error_code code = error_code::none; + std::size_t offset = 0; ///< byte offset of the offending character +}; + +/// FloatType: the number_float_t of the document, whose overflow parse() rejects +template +class builder +{ + public: + builder(document_data& d, const char* src, std::size_t size) noexcept + : doc(d) + , b(reinterpret_cast(src)) + , e(b + size) + {} + + /// returns false and fills `failure` on error + bool run() + { + cursor c(*this); + return c.run(); + } + + builder(const builder&) = delete; + builder& operator=(const builder&) = delete; + + parse_failure failure{}; + + private: + struct frame + { + std::uint32_t idx; + std::uint32_t count; + bool is_object; + }; + + document_data& doc; + const unsigned char* const b; + const unsigned char* const e; + + // the open array/object is in the cursor; enclosing ones on a stack that is + // inline for the first 64 levels + frame shallow[64]; + std::vector deep{}; + + NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept + { + failure.code = c; + failure.offset = static_cast(at - b); + doc.tape_size = 0; + return false; + } + + /// a failure (recorded by fail) as a comment() result + const unsigned char* fail_at(error_code c, const unsigned char* at) noexcept + { + fail(c, at); + return nullptr; + } + + /// a decoded string: p after its closing quote (nullptr: an error), and + /// its bytes in the arena + struct decoded + { + const unsigned char* p; + std::size_t start; + std::size_t len; + }; + + decoded failed(error_code c, const unsigned char* at) noexcept + { + fail(c, at); + return decoded{nullptr, 0, 0}; + } + + /// the comment at p (*p == '/'): the position after it, or nullptr on error + NLOHMANN_VIEW_NOINLINE const unsigned char* comment(const unsigned char* p) + { + if (e - p < 2) + { + ++p; + return fail_at(error_code::comment_start, p); + } + if (p[1] == '/') + { + p += 2; + while (p != e && *p != '\n' && *p != '\r' && !(NulIsEnd && *p == 0)) + { + ++p; + } + if (NulIsEnd && p != e && *p == 0) + { + ++p; // as in parse(), a null byte ends the comment like a line break + } + return p; + } + if (p[1] == '*') + { + p += 2; + for (;;) + { + if (p == e || (NulIsEnd && *p == 0)) + { + return fail_at(error_code::comment_unterminated, p); + } + if (*p == '*' && p + 1 != e && p[1] == '/') + { + p += 2; + return p; + } + ++p; + } + } + ++p; + return fail_at(error_code::comment_start, p); + } + + /// the index is full (n nodes, parsed up to at): extrapolate the node + /// count from the nodes per input byte so far (with headroom, and at least + /// 1.5 times as many), so that dense inputs regrow once instead of + /// doubling repeatedly; returns the new node array + NLOHMANN_VIEW_NOINLINE node* grow(std::size_t n, const unsigned char* at) + { + const std::uint64_t done = static_cast(at - b) + 1; + const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; + doc.tape_size = n; + doc.reserve((std::max)(static_cast(guess + guess / 4 + 64), n + n / 2 + 64)); + return doc.tape; + } + + /// four hex digits at p as a code unit (p moves past them), or -1 (p at + /// the first bad digit); one table lookup per digit and a single check, + /// as in yyjson's read_hex_u16 + NLOHMANN_VIEW_ALWAYS_INLINE int hex4(const unsigned char*& p) noexcept + { + // digit values; 0xF0: not a hex digit + static const std::uint8_t hex[256] = + { + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 10, 11, 12, 13, 14, 15, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 10, 11, 12, 13, 14, 15, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, 0xF0, + }; + if (NLOHMANN_VIEW_LIKELY(e - p >= 4)) + { + const unsigned d0 = hex[p[0]], d1 = hex[p[1]], d2 = hex[p[2]], d3 = hex[p[3]]; + if (NLOHMANN_VIEW_LIKELY(((d0 | d1 | d2 | d3) & 0xF0u) == 0)) + { + p += 4; + return static_cast((d0 << 12) | (d1 << 8) | (d2 << 4) | d3); + } + } + p = hex4_error(p); + return -1; + } + + /// hex4() failed: the first bad digit (none: p) + NLOHMANN_VIEW_NOINLINE const unsigned char* hex4_error(const unsigned char* p) noexcept + { + if (e - p >= 4) + { + while (is_hex(*p)) + { + ++p; + } + } + return p; + } + + /// escapes present (or an error) in the string at s, scanned up to p: + /// decode into the arena + NLOHMANN_VIEW_NOINLINE decoded slow_string(const unsigned char* s, const unsigned char* p) + { + // single-character escapes; 0: invalid (and 'u', handled separately) + static const char simple_escape[128] = + { + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, '"', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '/', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '\\', 0, 0, 0, + 0, 0, '\b', 0, 0, 0, '\f', 0, 0, 0, 0, 0, 0, 0, '\n', 0, 0, 0, '\r', 0, '\t', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + }; + const std::size_t start = arena_used(); + arena_run(s, static_cast(p - s)); + for (;;) + { + if (p == e) + { + return failed(error_code::string_missing_quote, p); + } + const unsigned char c = *p; + if (c == '"') + { + ++p; + return decoded{p, start, arena_used() - start}; + } + if (c != '\\') + { + if (NulIsEnd && c == 0) + { + return failed(error_code::string_missing_quote, p); + } + return failed(c < 0x20 ? error_code::string_control_character : error_code::string_utf8, p); + } + ++p; + if (p == e) + { + return failed(error_code::string_missing_quote, p); + } + const unsigned char d = *p++; + arena_ensure(4); + if (d == 'u') + { + int cp = hex4(p); + if (NLOHMANN_VIEW_UNLIKELY(cp < 0)) + { + return failed(error_code::string_unicode_hex, p); + } + if (NLOHMANN_VIEW_UNLIKELY((cp & 0xF800) == 0xD800)) // a surrogate + { + if (cp >= 0xDC00) + { + return failed(error_code::string_surrogate_low, p); + } + if (e - p < 2 || p[0] != '\\' || p[1] != 'u') + { + return failed(error_code::string_surrogate_high, p); + } + p += 2; + const int lo = hex4(p); + if (lo < 0) + { + return failed(error_code::string_unicode_hex, p); + } + if (lo < 0xDC00 || lo > 0xDFFF) + { + return failed(error_code::string_surrogate_high, p); + } + cp = 0x10000 + ((cp - 0xD800) << 10) + (lo - 0xDC00); + } + aw = put_utf8(aw, cp); + } + else if (NLOHMANN_VIEW_LIKELY(d < 128 && simple_escape[d] != 0)) + { + *aw++ = simple_escape[d]; + } + else + { + --p; + return failed(error_code::string_escape, p); + } + if (p != e && *p == '\\') + { + continue; // consecutive escapes ("\u00e4\u00f6"): no run in between + } + const unsigned char* const r = p; + p = scan_string_run(p, e); + arena_run(r, static_cast(p - r)); + } + } + + /// does the float token [s, p) overflow FloatType? (parse() rejects it) + NLOHMANN_VIEW_NOINLINE static bool float_overflows(const unsigned char* s, const unsigned char* p) + { + const auto* const first = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const auto* const last = reinterpret_cast(p); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const char* const dot = std::find(first, last, '.'); + const char* const exponent = std::find_if(first, last, [](char c) + { + return c == 'e' || c == 'E'; + }); + const auto v = convert_float(first, last, dot == last ? std::string::npos : static_cast(dot - first), + static_cast(exponent - first)); + return v > (std::numeric_limits::max)() || v < -(std::numeric_limits::max)(); + } + + /// does the magnitude digits [d, d + n) exceed the given limit (same length)? + static bool digits_exceed(const unsigned char* d, const char* limit, std::size_t n) noexcept + { + return std::memcmp(d, limit, n) > 0; + } + + static bool is_hex(unsigned char c) noexcept + { + return (c >= '0' && c <= '9') || (c >= 'A' && c <= 'F') || (c >= 'a' && c <= 'f'); + } + + // decode arena: the std::string in the document, written through a raw + // pointer (resized ahead in large steps; trimmed when parsing succeeds) + char* aw = nullptr; + char* aend = nullptr; + + NLOHMANN_VIEW_ALWAYS_INLINE std::size_t arena_used() const noexcept + { + return aw != nullptr ? static_cast(aw - doc.arena.data()) : 0; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void arena_ensure(std::size_t n) + { + if (NLOHMANN_VIEW_UNLIKELY(static_cast(aend - aw) < n)) + { + arena_grow(n); + } + } + + NLOHMANN_VIEW_NOINLINE void arena_grow(std::size_t n) + { + const std::size_t used = arena_used(); + doc.arena.resize((std::max)(doc.arena.size() * 2, used + n + 256)); + aw = &doc.arena[0] + used; + aend = &doc.arena[0] + doc.arena.size(); + } + + /// append the run [r, r + n) to the arena; short runs as one fixed-size + /// 16-byte move when both sides have the room (no library call) + NLOHMANN_VIEW_ALWAYS_INLINE void arena_run(const unsigned char* r, std::size_t n) + { + arena_ensure(n + 16); + if (n <= 16 && e - r >= 16) + { + std::memcpy(aw, r, 16); + } + else + { + std::memcpy(aw, r, n); + } + aw += n; + } + + /// UTF-8 encoding of cp at w (room for 4 bytes) + static char* put_utf8(char* w, int cp) noexcept + { + if (cp < 0x80) + { + *w++ = static_cast(cp); + } + else if (cp < 0x800) + { + *w++ = static_cast(0xC0 | (cp >> 6)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + else if (cp < 0x10000) + { + *w++ = static_cast(0xE0 | (cp >> 12)); + *w++ = static_cast(0x80 | ((cp >> 6) & 0x3F)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + else + { + *w++ = static_cast(0xF0 | (cp >> 18)); + *w++ = static_cast(0x80 | ((cp >> 12) & 0x3F)); + *w++ = static_cast(0x80 | ((cp >> 6) & 0x3F)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + return w; + } + + /// The parse state and the parser proper. The cursor is a local object of + /// run() whose address never escapes (everything it calls out of line is a + /// member of the builder and gets the positions it needs), so that the + /// compiler keeps the state in registers instead of reloading it from + /// memory after every node store and call. + struct cursor + { + explicit cursor(builder& owner) noexcept + : cold(owner) + , b(owner.b) + , p(owner.b) + , e(owner.e) + {} + + builder& cold; ///< out-of-line helpers and state that needs no registers + const unsigned char* const b; + const unsigned char* p; + const unsigned char* const e; + node* base = nullptr; + node* out = nullptr; + node* cap = nullptr; + + // the open array/object + std::uint32_t cur_idx = 0; + std::uint32_t cur_count = 0; + bool cur_is_object = false; + std::size_t depth = 0; + + NLOHMANN_VIEW_ALWAYS_INLINE bool run() + { + cold.doc.reserve(estimate_nodes(reinterpret_cast(b), static_cast(e - b))); + base = cold.doc.tape; + out = base; + cap = base + cold.doc.tape_cap; + + if (e - p >= 3 && p[0] == 0xEF && p[1] == 0xBB && p[2] == 0xBF) + { + p += 3; // byte order mark + } + if (!ws()) + { + return false; + } + if (p == e || (NulIsEnd && *p == 0)) + { + return fail(error_code::empty_input); + } + + // root value + switch (cur()) + { + case '{': + open(value_t::object); + ++p; + goto obj_first; + case '[': + open(value_t::array); + ++p; + goto arr_first; + default: + if (!scalar()) + { + return false; + } + goto root_done; + } + + // value dispatch, expanded once for array elements and once for member + // values: each jump has its own history (arrays tend to hold one kind of + // value), and the continuation needs no branch on the container kind +#define NLOHMANN_VIEW_VALUE(NEXT) \ + switch (cur()) \ + { \ + case '"': \ + if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \ + goto NEXT; \ + case '{': \ + open(value_t::object); \ + ++p; \ + goto obj_first; \ + case '[': \ + open(value_t::array); \ + ++p; \ + goto arr_first; \ + case '-': \ + if (NLOHMANN_VIEW_UNLIKELY(!number())) { return false; } \ + goto NEXT; \ + case '0': case '1': case '2': case '3': \ + case '4': case '5': case '6': case '7': case '8': case '9': \ + if (NLOHMANN_VIEW_UNLIKELY(!number())) { return false; } \ + goto NEXT; \ + case 't': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal("true", 4, value_t::boolean, node_flags::is_true))) { return false; } \ + goto NEXT; \ + case 'f': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal_false())) { return false; } \ + goto NEXT; \ + case 'n': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal("null", 4, value_t::null, 0))) { return false; } \ + goto NEXT; \ + default: \ + return fail(error_code::unexpected_value); \ + } + +arr_first: + if (!ws()) + { + return false; + } + if (cur() == ']') + { + ++p; + goto close_container; + } +value: + NLOHMANN_VIEW_VALUE(arr_next) +arr_next: + ++cur_count; + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ',')) + { + ++p; + if (!ws()) + { + return false; + } + if (TrailingCommas && cur() == ']') + { + ++p; + goto close_container; + } + goto value; + } + if (cur() == ']') + { + ++p; + goto close_container; + } + return fail(error_code::expected_array_end); + +obj_first: + if (!ws()) + { + return false; + } + if (cur() == '}') + { + ++p; + goto close_container; + } +obj_key: + if (NLOHMANN_VIEW_UNLIKELY(cur() != '"')) + { + return fail(error_code::expected_key); + } + if (NLOHMANN_VIEW_UNLIKELY(!string())) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ':' && (Sentinel || e - p >= 2) && p[1] == ' ')) + { + p += 2; // ": " (pretty-printed input; a fast path of yyjson) + } + else + { + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_UNLIKELY(cur() != ':')) + { + return fail(error_code::expected_colon); + } + ++p; + } + if (!ws()) + { + return false; + } + NLOHMANN_VIEW_VALUE(obj_next) +obj_next: + ++cur_count; + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ',')) + { + ++p; + if (!ws()) + { + return false; + } + if (TrailingCommas && cur() == '}') + { + ++p; + goto close_container; + } + goto obj_key; + } + if (cur() == '}') + { + ++p; + goto close_container; + } + return fail(error_code::expected_object_end); + +#undef NLOHMANN_VIEW_VALUE + +close_container: + close(); + if (NLOHMANN_VIEW_UNLIKELY(depth == 0)) + { + goto root_done; + } + if (cur_is_object) + { + goto obj_next; + } + goto arr_next; + +root_done: + if (!ws()) + { + return false; + } + if (p != e && !(NulIsEnd && *p == 0)) + { + return fail(error_code::trailing_characters); + } + cold.doc.tape_size = static_cast(out - base); + cold.doc.arena.resize(cold.arena_used()); + return true; + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool fail(error_code c) noexcept + { + return cold.fail(c, p); + } + + /// the current byte, or 0 at the end. With a NUL-terminated input + /// (Sentinel) the terminator is read instead of checking the bounds; a 0 + /// never matches a JSON token, so the error paths tell the end apart. + NLOHMANN_VIEW_ALWAYS_INLINE unsigned char cur() const noexcept + { + return Sentinel ? *p : (p != e ? *p : 0); + } + + /// root scalar + NLOHMANN_VIEW_ALWAYS_INLINE bool scalar() + { + switch (cur()) + { + case '"': + return string(); + case 't': + return literal("true", 4, value_t::boolean, node_flags::is_true); + case 'f': + return literal_false(); + case 'n': + return literal("null", 4, value_t::null, 0); + case '-': + return number(); + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + return number(); + default: + return fail(error_code::unexpected_value); + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool literal_false() + { + return literal("false", 5, value_t::boolean, 0); + } + + /// skip whitespace (and comments); false on a malformed comment + NLOHMANN_VIEW_ALWAYS_INLINE bool ws() + { + const unsigned char c = cur(); + if (NLOHMANN_VIEW_LIKELY(c > ' ' && (!Comments || c != '/'))) + { + return true; // no whitespace: the common case in minified input + } + return ws_slow(); + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool ws_slow() + { + for (;;) + { + if (cur() == ' ' && (Sentinel || e - p >= 2) && p[1] > ' ' && (!Comments || p[1] != '/')) + { + ++p; // single space, e.g. after ':' or ',' + return true; + } + if (cur() == '\n' || cur() == '\r') + { + // (a branch, not an add of the comparison: p must not + // wait for the byte after the line break) + if (NLOHMANN_VIEW_LIKELY(cur() == '\n')) + { + ++p; + } + else if ((Sentinel || e - p >= 2) && p[1] == '\n') + { + p += 2; + } + else + { + ++p; + } + // indentation: two spaces per step, fixed offsets (after yyjson) + while (e - p >= 32) + { +#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(load16(p + 2 * (i)) == 0x2020)) {} else { p += 2 * (i); goto indent_done; } + NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) +#undef NLOHMANN_VIEW_STEP + p += 32; + } +indent_done: + ; + } + for (unsigned char c = cur(); c == ' ' || c == '\n' || c == '\r' || c == '\t'; c = cur()) + { + ++p; + } + if (Comments && cur() == '/') + { + const unsigned char* const q = cold.comment(p); + if (q == nullptr) + { + return false; + } + p = q; + continue; + } + return true; + } + } + + /// append a node: (kind, flags, extra, off) and the second word (len, or + /// an integer's value); two stores on little-endian targets + NLOHMANN_VIEW_ALWAYS_INLINE node* emit(value_t k, std::uint8_t flags, std::uint16_t extra, std::size_t off, std::uint64_t second) + { + if (NLOHMANN_VIEW_UNLIKELY(out == cap)) + { + const std::size_t n = static_cast(out - base); + base = cold.grow(n, p); + out = base + n; + cap = base + cold.doc.tape_cap; + } + node* n = out++; +#if NLOHMANN_VIEW_LITTLE_ENDIAN + const std::uint64_t first = static_cast(k) | (static_cast(flags) << 8) + | (static_cast(extra) << 16) | (static_cast(off) << 32); + std::memcpy(reinterpret_cast(n), &first, 8); + std::memcpy(reinterpret_cast(n) + 8, &second, 8); +#else + n->kind = static_cast(k); + n->flags = flags; + n->extra = extra; + n->off = static_cast(off); + set_integer_bits(*n, second); +#endif + return n; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void open(value_t k) + { + const auto idx = static_cast(emit(k, 0, 0, static_cast(p - b), 0) - base); + if (depth != 0) + { + const frame f = {cur_idx, cur_count, cur_is_object}; + if (NLOHMANN_VIEW_LIKELY(depth <= 64)) + { + cold.shallow[depth - 1] = f; + } + else + { + cold.deep.push_back(f); + } + } + ++depth; + cur_idx = idx; + cur_count = 0; + cur_is_object = k == value_t::object; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void close() + { + node& n = base[cur_idx]; + n.len = cur_count; + n.next = static_cast(out - base) - cur_idx; + if (--depth != 0) + { + frame f; + if (NLOHMANN_VIEW_LIKELY(depth <= 64)) + { + f = cold.shallow[depth - 1]; + } + else + { + f = cold.deep.back(); + cold.deep.pop_back(); + } + cur_idx = f.idx; + cur_count = f.count; + cur_is_object = f.is_object; + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool literal(const char* text, std::size_t n, value_t k, std::uint8_t flags) + { + if (NLOHMANN_VIEW_UNLIKELY(e - p < static_cast(n) || std::memcmp(p, text, n) != 0)) + { + return fail(error_code::invalid_literal); + } + emit(k, flags, 0, static_cast(p - b), n); + p += n; + return true; + } + + /// a number at p; the sign is known from the dispatch (so that p does + /// not have to wait for the first byte) + template + NLOHMANN_VIEW_ALWAYS_INLINE bool number() + { + const unsigned char* const s = p; + if (negative) + { + ++p; + } + const unsigned char* const int_start = p; + if (p != e && *p == '0') + { + ++p; + } + else if (NLOHMANN_VIEW_LIKELY(p != e && *p >= '1' && *p <= '9')) + { + p = skip_digits(p + 1, e); + } + else + { + return fail(error_code::number_after_minus); + } + const std::size_t int_digits = static_cast(p - int_start); + std::size_t frac_digits = 0; + bool is_float = false; + if (p != e && *p == '.') + { + ++p; + const unsigned char* const f0 = p; + p = skip_digits(p, e); + if (NLOHMANN_VIEW_UNLIKELY(p == f0)) + { + return fail(error_code::number_after_dot); + } + frac_digits = static_cast(p - f0); + is_float = true; + } + long exponent = 0; + if (p != e && (*p | 0x20) == 'e') + { + ++p; + bool exp_negative = false; + if (p != e && (*p == '+' || *p == '-')) + { + exp_negative = *p == '-'; + ++p; + } + if (NLOHMANN_VIEW_UNLIKELY(p == e || !is_digit(*p))) + { + return fail(error_code::number_after_exponent); + } + while (p != e && is_digit(*p)) + { + if (exponent < 100000) + { + exponent = exponent * 10 + (*p - '0'); + } + ++p; + } + if (exp_negative) + { + exponent = -exponent; + } + is_float = true; + } + + value_t kind = is_float ? value_t::number_float : (negative ? value_t::number_integer : value_t::number_unsigned); + if (!is_float) + { + // integers that do not fit become floats, as in parse() + if (NLOHMANN_VIEW_UNLIKELY(int_digits >= 19)) + { + if (negative) + { + if (int_digits > 19 || (int_digits == 19 && digits_exceed(int_start, "9223372036854775808", 19))) + { + kind = value_t::number_float; + } + } + else if (int_digits > 20 || (int_digits == 20 && digits_exceed(int_start, "18446744073709551615", 20))) + { + kind = value_t::number_float; + } + } + } + // parse() rejects floats that overflow; only numbers whose magnitude + // could reach the largest FloatType (1e308 for double, 1e38 for + // float) need the conversion + if (NLOHMANN_VIEW_UNLIKELY(static_cast(int_digits) + exponent > std::numeric_limits::max_exponent10 - 8 && kind == value_t::number_float)) + { + if (cold.float_overflows(s, p)) + { + p = s; + return fail(error_code::number_overflow); + } + } + const auto layout = static_cast((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8)); + std::uint64_t second = static_cast(p - s); + if (kind != value_t::number_float) + { + // integers are converted now, while their digits are in cache + const std::uint64_t m = int_digits <= 19 ? parse_upto19(int_start, static_cast(int_digits), e) + : parse_upto19(int_start, 19, e) * 10 + static_cast(int_start[19] - '0'); + second = negative ? 0 - m : m; + } + emit(kind, 0, layout, static_cast(s - b), second); + return true; + } + + /// a string (value or key) at p + NLOHMANN_VIEW_ALWAYS_INLINE bool string() + { + ++p; // opening quote + const unsigned char* const s = p; + p = scan_string_run(p, e); + if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"')) + { + emit(value_t::string, 0, 0, static_cast(s - b), static_cast(p - s)); + ++p; + return true; + } + const decoded r = cold.slow_string(s, p); + if (r.p == nullptr) + { + return false; + } + p = r.p; + emit(value_t::string, node_flags::escaped, 0, r.start, r.len); + return true; + } + }; +}; + +/// run the builder with compile-time options +template +inline bool build_with(document_data& d, const char* src, std::size_t size, bool sentinel, parse_failure& failure) +{ + if (sentinel) + { + builder bld(d, src, size); + const bool ok = bld.run(); + failure = bld.failure; + return ok; + } + builder bld(d, src, size); + const bool ok = bld.run(); + failure = bld.failure; + return ok; +} + +/// sentinel: src[size] is readable and 0 (e.g. std::string); FloatType: the +/// number_float_t of the document +template +inline bool build(document_data& d, const char* src, std::size_t size, bool comments, bool trailing_commas, bool sentinel, parse_failure& failure) +{ + if (comments) + { + return trailing_commas ? build_with(d, src, size, sentinel, failure) + : build_with(d, src, size, sentinel, failure); + } + return trailing_commas ? build_with(d, src, size, sentinel, failure) + : build_with(d, src, size, sentinel, failure); +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp new file mode 100644 index 000000000..9d4223327 --- /dev/null +++ b/tests/src/unit-json_view_builder.cpp @@ -0,0 +1,327 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +#include +using nlohmann::json; + +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace +{ +using nlohmann::detail::view::document_data; +using nlohmann::detail::view::node; + +// the node index of a text; a vector input has no terminating NUL, so that +// AddressSanitizer catches any read past the last byte +struct built +{ + std::unique_ptr data{}; // NOLINT(readability-redundant-member-init) + std::vector copy{}; // NOLINT(readability-redundant-member-init) + bool ok = false; + nlohmann::detail::view::parse_failure failure{}; +}; + +template +built build(const std::string& text, bool comments, bool trailing_commas, bool sentinel) +{ + built r; + r.data.reset(document_data::create(nlohmann::detail::view::estimate_nodes(text.data(), text.size()))); + const char* src = text.c_str(); + if (!sentinel) + { + r.copy.assign(text.begin(), text.end()); + src = r.copy.data(); + } + r.ok = nlohmann::detail::view::build < FloatType, !nlohmann::detail::abi_config::strict_nul_handling > (*r.data, src, text.size(), comments, trailing_commas, sentinel, r.failure); + r.data->src = src; + r.data->base[0] = src; + r.data->base[1] = r.data->arena.data(); + return r; +} + +// the value of a subtree, as json::parse would build it +json value_of(const document_data& d, const node*& n) +{ + const node& x = *n; + ++n; + switch (static_cast(x.kind)) + { + case json::value_t::object: + { + json o = json::object(); + const node* const end = &x + x.next; + while (n != end) + { + const std::string key(d.str(*n), n->len); + ++n; + o[key] = value_of(d, n); + } + return o; + } + case json::value_t::array: + { + json a = json::array(); + const node* const end = &x + x.next; + while (n != end) + { + a.push_back(value_of(d, n)); + } + return a; + } + case json::value_t::string: + return std::string(d.str(x), x.len); + case json::value_t::boolean: + return (x.flags & nlohmann::detail::view::node_flags::is_true) != 0; + case json::value_t::number_integer: + return static_cast(nlohmann::detail::view::integer_bits(x)); + case json::value_t::number_unsigned: + return nlohmann::detail::view::integer_bits(x); + case json::value_t::number_float: + return json::parse(std::string(d.src + x.off, x.len)).get(); + case json::value_t::null: + case json::value_t::binary: + case json::value_t::discarded: + default: + return nullptr; + } +} + +json value_of(const built& b) +{ + const node* n = b.data->tape; + json v = value_of(*b.data, n); + CHECK(n == b.data->tape + b.data->tape_size); + return v; +} + +// accept/reject and the value must match json::parse, for all options and +// with and without a NUL after the text +void check_same(const std::string& text) +{ + CAPTURE(text); + for (int options = 0; options < 4; ++options) + { + const bool comments = (options & 1) != 0; + const bool trailing_commas = (options & 2) != 0; + const bool accepted = json::accept(text, comments, trailing_commas); + for (const bool sentinel : + { + true, false + }) + { + const built b = build(text, comments, trailing_commas, sentinel); + CHECK(b.ok == accepted); + if (b.ok && accepted) + { + CHECK(value_of(b) == json::parse(text, nullptr, true, comments, trailing_commas)); + } + } + } +} + +// a small deterministic generator of documents +struct generator +{ + std::mt19937 rng{5295}; + + int r(int n) + { + return static_cast(rng() % static_cast(n)); + } + + void ws(std::string& o) + { + for (int n = r(4) == 0 ? r(12) : r(2); n > 0; --n) + { + o += " \n\t\r "[r(6)]; + } + } + + void str(std::string& o) + { + static const char* const pieces[] = {"a", "Z", " ", "~", "\\n", "\\\"", "\\\\", "\\/", "\\u00e9", "\\ud83d\\ude00", "\xc3\xa9", "\xe3\x81\x82", "\xf0\x9f\x98\x80", "\x7f", "\\u001f", "long enough text to leave the first 16 bytes"}; + o += '"'; + for (int n = r(3) == 0 ? r(20) : r(6); n > 0; --n) + { + o += pieces[r(16)]; + } + o += '"'; + } + + void num(std::string& o) + { + static const char* const numbers[] = {"0", "-0", "1", "-1", "12", "123456789", "1234567890123456789", "9223372036854775807", "-9223372036854775808", + "9223372036854775808", "18446744073709551615", "18446744073709551616", "-9223372036854775809", + "1.5", "-2.25e-3", "1e10", "1E+2", "0.000001", "3.141592653589793238462643", "1e308", "-1e-400", "123.456e7" + }; + o += numbers[r(22)]; + } + + void value(std::string& o, int depth) + { + ws(o); + const int k = depth > 5 ? 2 + r(6) : r(8); + if (k == 0 || k == 1) + { + const bool object = k == 0; + o += object ? '{' : '['; + for (int i = r(5); i > 0; --i) + { + ws(o); + if (object) + { + str(o); + ws(o); + o += ':'; + } + value(o, depth + 1); + o += i > 1 ? "," : ""; + } + ws(o); + o += object ? '}' : ']'; + } + else if (k < 4) + { + str(o); + } + else if (k < 6) + { + num(o); + } + else + { + static const char* const literals[] = {"true", "false", "null"}; + o += literals[r(3)]; + } + ws(o); + } +}; +} // namespace + +TEST_CASE("json_view builder") +{ + SECTION("scalars and containers") + { + for (const char* text : + { + "null", "true", "false", "0", "-0", "42", "-42", "1.5", "\"\"", "\"abc\"", "[]", "{}", "[1,2,3]", "{\"a\":1,\"b\":[true,null]}", + " [ 1 , 2 ] ", "{\"a\" : {\"b\" : {}}}", "[[[]]]", "\"\\u00e4\\n\\ud83d\\ude00\"", "{\"a\":1,\"a\":2}", "18446744073709551616", + "-9223372036854775809", "123456789012345678901234567890", "1e400", "-1e400", "1.7976931348623157e308" + }) + { + check_same(text); + } + } + + SECTION("malformed input") + { + for (const char* text : + { + "", " ", "[", "]", "{", "}", "[1,]", "{\"a\":1,}", "[1 2]", "{\"a\" 1}", "{1:2}", "tru", "nul", "fals", "truex", "-", "01", "1.", ".5", "1e", "1e+", + "\"", "\"abc", "\"\\x\"", "\"\\u12\"", "\"\\u12G4\"", "\"\\ud800\"", "\"\\udc00\"", "\"\\ud800\\u0041\"", "\"\x01\"", "\"\xff\"", "\"\xc3\"", + "\"\xe0\x80\x80\"", "\"\xed\xa0\x80\"", "[1]x", "[1] [2]", "/", "/*", "/* */ 1", "// c\n1", "1 // c", "[1,/*c*/2]", "[1,2,]" + }) + { + check_same(text); + } + } + + SECTION("NUL, BOM, and whitespace") + { + check_same(std::string("[1]\0garbage", 11)); + check_same(std::string("[1\0]", 4)); + check_same(std::string("[1, // c\0\n2]", 12)); + check_same(std::string("[1, /* c\0 */ 2]", 15)); + check_same("\xEF\xBB\xBF[1]"); + check_same("\xEF\xBB[1]"); + check_same(" \t\r\n 7 \n"); + for (const char* text : + {"[1]\r", "[1]\n", "[1]\r\n", "[1,\r2]", "[1,\r\n2]", "7\r", "\"x\"\r", "{\"a\":\r\n1}\r", "[\n 1,\n 2\n]", "{\n \"a\": [\n 1\n ]\n}" + }) + { + check_same(text); + } + } + + SECTION("deep nesting") + { + // the open containers beyond 64 levels live on the heap + for (const std::size_t depth : + { + 63u, 64u, 65u, 1000u, 100000u + }) + { + const std::string arrays = std::string(depth, '[') + std::string(depth, ']'); + const built b = build(arrays, false, false, false); + REQUIRE(b.ok); + CHECK(b.data->tape_size == depth); + CHECK(b.data->tape[0].next == depth); + std::string objects; + for (std::size_t i = 0; i < depth; ++i) + { + objects += "{\"a\":"; + } + objects += '1' + std::string(depth, '}'); + const built o = build(objects, false, false, false); + REQUIRE(o.ok); + CHECK(o.data->tape_size == (2 * depth) + 1); + CHECK(!build(std::string(depth, '[') + std::string(depth - 1, ']'), false, false, false).ok); + } + } + + SECTION("generated documents and damaged copies") + { + generator g; + for (int i = 0; i < 3000; ++i) + { + std::string text; + g.value(text, 0); + check_same(text); + // damage: flip one byte, or cut the text + std::string damaged = text; + const auto at = static_cast(g.r(static_cast(damaged.size()))); + static const char replacements[] = {'x', '"', '\\', ',', ':', ']', '}', '[', '{', '1', '-', '.', 'e', '\0', '\n', '/'}; + damaged[at] = replacements[g.r(16)]; + check_same(damaged); + check_same(text.substr(0, at)); + } + } + + SECTION("test files") + { + for (const char* name : + { + "/json.org/1.json", "/json.org/2.json", "/json.org/3.json", "/json.org/4.json", "/json.org/5.json", + "/json_testsuite/sample.json", "/nativejson-benchmark/canada.json", "/nativejson-benchmark/citm_catalog.json", + "/nativejson-benchmark/twitter.json", "/json_tests/pass1.json", "/json_tests/pass2.json", "/json_tests/pass3.json" + }) + { + CAPTURE(name); + std::ifstream f(std::string(TEST_DATA_DIRECTORY) + name, std::ios::binary); + std::stringstream ss; + ss << f.rdbuf(); + const std::string text = ss.str(); + REQUIRE(!text.empty()); + const built b = build(text, false, false, true); + REQUIRE(b.ok); + CHECK(value_of(b) == json::parse(text)); + } + } +}