Files
json/include/nlohmann/detail/view/builder.hpp
Niels Lohmann 904c8c6710 Index the large objects of json_view
Lookups in objects are linear, as for ordered_json. Objects with 128
members or more now get a hash table after parsing (open addressing; the
first of duplicate keys is kept, as for the linear search), so that
operator[], at(), find(), contains(), count(), value(), and JSON pointers
take constant time on average in them; the idea of switching to a hash
table for large objects is Boost.JSON's. The parser notes such objects when
it closes them (out of line, so that the parse loop only has a call for
it), and the object node keeps the number of its table.

Looking up each key of an object with 10,000 members: 59.8 ms -> 0.16 ms.
Parsing (json_document::parse, best of 7, separate processes): most files
within 1%; canada +5%, mesh.pretty +3%, citm +3%.

Tests: objects with 127, 128, 129, and 10,000 members (escaped, empty,
and duplicate keys, missing keys, comparisons), nested large objects, and
documents reused with read().

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-30 21:01:58 +02:00

1054 lines
37 KiB
C++

// __ _____ _____ _____
// __| | __| | | | JSON for Modern C++
// | | |__ | | | | | | version 3.12.0
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
//
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
// SPDX-FileCopyrightText: 2020 YaoYuan <https://github.com/ibireme/yyjson>
// SPDX-License-Identifier: MIT
#pragma once
#include <algorithm> // find, find_if, max
#include <array> // array
#include <cstddef> // size_t, ptrdiff_t
#include <cstdint> // int64_t, uint8_t, uint16_t, uint32_t, uint64_t
#include <cstring> // memcmp, memcpy
#include <limits> // numeric_limits
#include <string> // string
#include <vector> // vector
#include <nlohmann/json.hpp>
#include <nlohmann/detail/view/document_data.hpp>
#include <nlohmann/detail/view/macro_scope.hpp>
#include <nlohmann/detail/view/node.hpp>
#include <nlohmann/detail/view/scan.hpp>
// The view's parser: one pass over the input that emits the node index (see
// node.hpp). The table-driven decoding of \u escapes and the fast paths for
// ": " and indentation follow yyjson (https://github.com/ibireme/yyjson, MIT
// license).
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
{
namespace view
{
enum class error_code : std::uint8_t
{
none,
empty_input,
unexpected_value,
invalid_literal,
expected_key,
expected_colon,
expected_array_end,
expected_object_end,
trailing_characters,
number_after_minus,
number_after_dot,
number_after_exponent,
number_overflow,
string_missing_quote,
string_control_character,
string_utf8,
string_escape,
string_unicode_hex,
string_surrogate_high,
string_surrogate_low,
comment_start,
comment_unterminated,
input_too_large,
};
struct parse_failure
{
error_code code = error_code::none;
std::size_t offset = 0; ///< byte offset of the offending character
};
/// FloatType: the number_float_t of the document, whose overflow parse() rejects
template<typename FloatType, bool Comments, bool TrailingCommas, bool NulIsEnd, bool Sentinel>
class builder
{
public:
builder(document_data& d, const char* src, std::size_t size) noexcept
: doc(d)
, b(reinterpret_cast<const unsigned char*>(src))
, e(b + size)
{}
/// returns false and fills `failure` on error
bool run()
{
cursor c(*this);
return c.run();
}
builder(const builder&) = delete;
builder& operator=(const builder&) = delete;
builder(builder&&) = delete;
builder& operator=(builder&&) = delete;
~builder() = default;
/// where and why the parse failed (after run() returned false)
const parse_failure& failure() const noexcept
{
return m_failure;
}
private:
struct frame
{
std::uint32_t idx;
std::uint32_t count;
bool is_object;
};
document_data& doc;
const unsigned char* const b;
const unsigned char* const e;
parse_failure m_failure{};
// the open array/object is in the cursor; enclosing ones on a stack that is
// inline for the first 64 levels
frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open
std::vector<frame> deep{};
/// remember an object to index after parsing (out of line, so that the
/// parse loop only has a call for it)
NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx)
{
doc.large_objects.push_back(idx);
}
NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept
{
m_failure.code = c;
m_failure.offset = static_cast<std::size_t>(at - b);
doc.tape_size = 0;
return false;
}
/// a failure (recorded by fail) as a comment() result
const unsigned char* fail_at(error_code c, const unsigned char* at) noexcept
{
fail(c, at);
return nullptr;
}
/// a decoded string: p after its closing quote (nullptr: an error), and
/// its bytes in the arena
struct decoded
{
const unsigned char* p;
std::size_t start;
std::size_t len;
};
decoded failed(error_code c, const unsigned char* at) noexcept
{
fail(c, at);
return decoded{nullptr, 0, 0};
}
/// the comment at p (*p == '/'): the position after it, or nullptr on error
NLOHMANN_VIEW_NOINLINE const unsigned char* comment(const unsigned char* p)
{
if (e - p < 2)
{
++p;
return fail_at(error_code::comment_start, p);
}
if (p[1] == '/')
{
p += 2;
while (p != e && *p != '\n' && *p != '\r' && !(NulIsEnd && *p == 0))
{
++p;
}
if (NulIsEnd && p != e && *p == 0)
{
++p; // as in parse(), a null byte ends the comment like a line break
}
return p;
}
if (p[1] == '*')
{
p += 2;
for (;;)
{
if (p == e || (NulIsEnd && *p == 0))
{
return fail_at(error_code::comment_unterminated, p);
}
if (*p == '*' && p + 1 != e && p[1] == '/')
{
p += 2;
return p;
}
++p;
}
}
++p;
return fail_at(error_code::comment_start, p);
}
/// the index is full (n nodes, parsed up to at): extrapolate the node
/// count from the nodes per input byte so far (with headroom, and at least
/// 1.5 times as many), so that dense inputs regrow once instead of
/// doubling repeatedly; returns the new node array
NLOHMANN_VIEW_NOINLINE node* grow(std::size_t n, const unsigned char* at)
{
const std::uint64_t done = static_cast<std::uint64_t>(at - b) + 1;
const std::uint64_t guess = static_cast<std::uint64_t>(n) * static_cast<std::uint64_t>(e - b + 1) / done;
const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t
doc.tape_size = n;
doc.reserve((std::max)(static_cast<std::size_t>(grown), n + (n / 2) + 64));
return doc.tape;
}
/// four hex digits at p as a code unit (p moves past them), or -1 (p at
/// the first bad digit); one table lookup per digit and a single check,
/// the four hex digits of a unicode escape (the library's table, after
/// yyjson's read_hex_u16), or -1
NLOHMANN_VIEW_ALWAYS_INLINE int hex4(const unsigned char*& p) noexcept
{
if (NLOHMANN_VIEW_LIKELY(e - p >= 4))
{
const int cp = hex_codepoint(p);
if (NLOHMANN_VIEW_LIKELY(cp >= 0))
{
p += 4;
return cp;
}
}
p = hex4_error(p);
return -1;
}
/// hex4() failed: the first bad digit (none: p)
NLOHMANN_VIEW_NOINLINE const unsigned char* hex4_error(const unsigned char* p) noexcept
{
if (e - p >= 4)
{
while (is_hex(*p))
{
++p;
}
}
return p;
}
/// escapes present (or an error) in the string at s, scanned up to p:
/// decode into the arena
NLOHMANN_VIEW_NOINLINE decoded slow_string(const unsigned char* s, const unsigned char* p)
{
// single-character escapes; 0: invalid (and 'u', handled separately)
static const std::array<char, 128> simple_escape =
{
{
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, '"', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '/', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '\\', 0, 0, 0,
0, 0, '\b', 0, 0, 0, '\f', 0, 0, 0, 0, 0, 0, 0, '\n', 0, 0, 0, '\r', 0, '\t', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0
}
};
const std::size_t start = arena_used();
arena_run(s, static_cast<std::size_t>(p - s));
for (;;)
{
if (p == e)
{
return failed(error_code::string_missing_quote, p);
}
const unsigned char c = *p;
if (c == '"')
{
++p;
return decoded{p, start, arena_used() - start};
}
if (c != '\\')
{
// (a NUL before the end of the input is a control character, as
// for json::parse, also where a NUL ends the input between values)
return failed(c < 0x20 ? error_code::string_control_character : error_code::string_utf8, p);
}
++p;
if (p == e)
{
return failed(error_code::string_missing_quote, p);
}
const unsigned char d = *p++;
arena_ensure(4);
if (d == 'u')
{
int cp = hex4(p);
if (NLOHMANN_VIEW_UNLIKELY(cp < 0))
{
return failed(error_code::string_unicode_hex, p);
}
if (NLOHMANN_VIEW_UNLIKELY((cp & 0xF800) == 0xD800)) // a surrogate
{
if (cp >= 0xDC00)
{
return failed(error_code::string_surrogate_low, p);
}
if (e - p < 2 || p[0] != '\\' || p[1] != 'u')
{
return failed(error_code::string_surrogate_high, p);
}
p += 2;
const int lo = hex4(p);
if (lo < 0)
{
return failed(error_code::string_unicode_hex, p);
}
if (lo < 0xDC00 || lo > 0xDFFF)
{
return failed(error_code::string_surrogate_high, p);
}
cp = 0x10000 + ((cp - 0xD800) << 10) + (lo - 0xDC00);
}
aw = put_utf8(aw, cp);
}
else if (NLOHMANN_VIEW_LIKELY(d < 128 && simple_escape[d] != 0))
{
*aw++ = simple_escape[d];
}
else
{
--p;
return failed(error_code::string_escape, p);
}
if (p != e && *p == '\\')
{
continue; // consecutive escapes ("\u00e4\u00f6"): no run in between
}
const unsigned char* const r = p;
p = scan_string_run(p, e);
arena_run(r, static_cast<std::size_t>(p - r));
}
}
/// does the float token [s, p) overflow FloatType? (parse() rejects it)
NLOHMANN_VIEW_NOINLINE static bool float_overflows(const unsigned char* s, const unsigned char* p)
{
const auto* const first = reinterpret_cast<const char*>(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
const auto* const last = reinterpret_cast<const char*>(p); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
const char* const dot = std::find(first, last, '.');
const char* const exponent = std::find_if(first, last, [](char c)
{
return c == 'e' || c == 'E';
});
const auto v = convert_float<FloatType>(first, last, dot == last ? std::string::npos : static_cast<std::size_t>(dot - first),
static_cast<std::size_t>(exponent - first));
return v > (std::numeric_limits<FloatType>::max)() || v < -(std::numeric_limits<FloatType>::max)();
}
/// does the magnitude digits [d, d + n) exceed the given limit (same length)?
static bool digits_exceed(const unsigned char* d, const char* limit, std::size_t n) noexcept
{
return std::memcmp(d, limit, n) > 0;
}
static bool is_hex(unsigned char c) noexcept
{
return (c >= '0' && c <= '9') || (c >= 'A' && c <= 'F') || (c >= 'a' && c <= 'f');
}
// decode arena: the std::string in the document, written through a raw
// pointer (resized ahead in large steps; trimmed when parsing succeeds)
char* aw = nullptr;
char* aend = nullptr;
NLOHMANN_VIEW_ALWAYS_INLINE std::size_t arena_used() const noexcept
{
return aw != nullptr ? static_cast<std::size_t>(aw - doc.arena.data()) : 0;
}
NLOHMANN_VIEW_ALWAYS_INLINE void arena_ensure(std::size_t n)
{
if (NLOHMANN_VIEW_UNLIKELY(static_cast<std::size_t>(aend - aw) < n))
{
arena_grow(n);
}
}
NLOHMANN_VIEW_NOINLINE void arena_grow(std::size_t n)
{
const std::size_t used = arena_used();
doc.arena.resize((std::max)(doc.arena.size() * 2, used + n + 256));
aw = &doc.arena[0] + used; // NOLINT(readability-container-data-pointer): data() is const before C++17
aend = &doc.arena[0] + doc.arena.size(); // NOLINT(readability-container-data-pointer)
}
/// append the run [r, r + n) to the arena; short runs as one fixed-size
/// 16-byte move when both sides have the room (no library call)
NLOHMANN_VIEW_ALWAYS_INLINE void arena_run(const unsigned char* r, std::size_t n)
{
arena_ensure(n + 16);
if (n <= 16 && e - r >= 16)
{
std::memcpy(aw, r, 16);
}
else
{
std::memcpy(aw, r, n);
}
aw += n;
}
/// UTF-8 encoding of cp at w (room for 4 bytes)
static char* put_utf8(char* w, int cp) noexcept
{
if (cp < 0x80)
{
*w++ = static_cast<char>(cp);
}
else if (cp < 0x800)
{
*w++ = static_cast<char>(0xC0 | (cp >> 6));
*w++ = static_cast<char>(0x80 | (cp & 0x3F));
}
else if (cp < 0x10000)
{
*w++ = static_cast<char>(0xE0 | (cp >> 12));
*w++ = static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
*w++ = static_cast<char>(0x80 | (cp & 0x3F));
}
else
{
*w++ = static_cast<char>(0xF0 | (cp >> 18));
*w++ = static_cast<char>(0x80 | ((cp >> 12) & 0x3F));
*w++ = static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
*w++ = static_cast<char>(0x80 | (cp & 0x3F));
}
return w;
}
/// a compile-time option as a runtime condition: testing the template
/// argument directly makes a condition like `TrailingCommas && c == ']'`
/// constant when the option is off, which MSVC reports as C4127
static NLOHMANN_VIEW_ALWAYS_INLINE bool enabled(bool option) noexcept
{
return option;
}
/// The parse state and the parser proper. The cursor is a local object of
/// run() whose address never escapes (everything it calls out of line is a
/// member of the builder and gets the positions it needs), so that the
/// compiler keeps the state in registers instead of reloading it from
/// memory after every node store and call.
struct cursor
{
explicit cursor(builder& owner) noexcept
: cold(owner)
, b(owner.b)
, p(owner.b)
, e(owner.e)
{}
builder& cold; ///< out-of-line helpers and state that needs no registers
const unsigned char* const b;
const unsigned char* p;
const unsigned char* const e;
node* base = nullptr;
node* out = nullptr;
node* cap = nullptr;
// the open array/object
std::uint32_t cur_idx = 0;
std::uint32_t cur_count = 0;
bool cur_is_object = false;
std::size_t depth = 0;
NLOHMANN_VIEW_ALWAYS_INLINE bool run()
{
cold.doc.reserve(estimate_nodes(reinterpret_cast<const char*>(b), static_cast<std::size_t>(e - b)));
base = cold.doc.tape;
out = base;
cap = base + cold.doc.tape_cap;
if (e - p >= 3 && p[0] == 0xEF && p[1] == 0xBB && p[2] == 0xBF)
{
p += 3; // byte order mark
}
if (!ws())
{
return false;
}
if (p == e || (NulIsEnd && *p == 0))
{
return fail(error_code::empty_input);
}
// root value
switch (cur())
{
case '{':
open(value_t::object);
++p;
goto obj_first;
case '[':
open(value_t::array);
++p;
goto arr_first;
default:
if (!scalar())
{
return false;
}
goto root_done;
}
// value dispatch, expanded once for array elements and once for member
// values: each jump has its own history (arrays tend to hold one kind of
// value), and the continuation needs no branch on the container kind
#define NLOHMANN_VIEW_VALUE(NEXT) \
switch (cur()) \
{ \
case '"': \
if (NLOHMANN_VIEW_UNLIKELY(!string<true>())) { return false; } \
goto NEXT; \
case '{': \
open(value_t::object); \
++p; \
goto obj_first; \
case '[': \
open(value_t::array); \
++p; \
goto arr_first; \
case '-': \
if (NLOHMANN_VIEW_UNLIKELY(!number<true>())) { return false; } \
goto NEXT; \
case '0': case '1': case '2': case '3': \
case '4': case '5': case '6': case '7': case '8': case '9': \
if (NLOHMANN_VIEW_UNLIKELY(!number<false>())) { return false; } \
goto NEXT; \
case 't': \
if (NLOHMANN_VIEW_UNLIKELY(!literal("true", 4, value_t::boolean, node_flags::is_true))) { return false; } \
goto NEXT; \
case 'f': \
if (NLOHMANN_VIEW_UNLIKELY(!literal_false())) { return false; } \
goto NEXT; \
case 'n': \
if (NLOHMANN_VIEW_UNLIKELY(!literal("null", 4, value_t::null, 0))) { return false; } \
goto NEXT; \
default: \
return fail(error_code::unexpected_value); \
}
arr_first:
if (!ws())
{
return false;
}
if (cur() == ']')
{
++p;
goto close_container;
}
value:
NLOHMANN_VIEW_VALUE(arr_next)
arr_next:
++cur_count;
if (!ws())
{
return false;
}
if (NLOHMANN_VIEW_LIKELY(cur() == ','))
{
++p;
if (!ws())
{
return false;
}
if (enabled(TrailingCommas) && cur() == ']')
{
++p;
goto close_container;
}
goto value;
}
if (cur() == ']')
{
++p;
goto close_container;
}
return fail(error_code::expected_array_end);
obj_first:
if (!ws())
{
return false;
}
if (cur() == '}')
{
++p;
goto close_container;
}
obj_key:
if (NLOHMANN_VIEW_UNLIKELY(cur() != '"'))
{
return fail(error_code::expected_key);
}
if (NLOHMANN_VIEW_UNLIKELY(!string<false>()))
{
return false;
}
if (NLOHMANN_VIEW_LIKELY(cur() == ':' && (Sentinel || e - p >= 2) && p[1] == ' '))
{
p += 2; // ": " (pretty-printed input; a fast path of yyjson)
}
else
{
if (!ws())
{
return false;
}
if (NLOHMANN_VIEW_UNLIKELY(cur() != ':'))
{
return fail(error_code::expected_colon);
}
++p;
}
if (!ws())
{
return false;
}
NLOHMANN_VIEW_VALUE(obj_next)
obj_next:
++cur_count;
if (!ws())
{
return false;
}
if (NLOHMANN_VIEW_LIKELY(cur() == ','))
{
++p;
if (!ws())
{
return false;
}
if (enabled(TrailingCommas) && cur() == '}')
{
++p;
goto close_object;
}
goto obj_key;
}
if (cur() == '}')
{
++p;
goto close_object;
}
return fail(error_code::expected_object_end);
#undef NLOHMANN_VIEW_VALUE
close_object:
// a large object gets a hash index (objects only, so that closing
// an array pays nothing for this)
if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members))
{
cold.note_large_object(cur_idx);
}
close_container:
close();
if (NLOHMANN_VIEW_UNLIKELY(depth == 0))
{
goto root_done;
}
if (cur_is_object)
{
goto obj_next;
}
goto arr_next;
root_done:
if (!ws())
{
return false;
}
if (p != e && !(NulIsEnd && *p == 0))
{
return fail(error_code::trailing_characters);
}
cold.doc.tape_size = static_cast<std::size_t>(out - base);
cold.doc.arena.resize(cold.arena_used());
return true;
}
NLOHMANN_VIEW_ALWAYS_INLINE bool fail(error_code c) noexcept
{
return cold.fail(c, p);
}
/// the current byte, or 0 at the end. With a NUL-terminated input
/// (Sentinel) the terminator is read instead of checking the bounds; a 0
/// never matches a JSON token, so the error paths tell the end apart.
NLOHMANN_VIEW_ALWAYS_INLINE unsigned char cur() const noexcept
{
if (Sentinel)
{
return *p;
}
return p != e ? *p : 0;
}
/// root scalar
NLOHMANN_VIEW_ALWAYS_INLINE bool scalar()
{
switch (cur())
{
case '"':
return string<true>();
case 't':
return literal("true", 4, value_t::boolean, node_flags::is_true);
case 'f':
return literal_false();
case 'n':
return literal("null", 4, value_t::null, 0);
case '-':
return number<true>();
case '0':
case '1':
case '2':
case '3':
case '4':
case '5':
case '6':
case '7':
case '8':
case '9':
return number<false>();
default:
return fail(error_code::unexpected_value);
}
}
NLOHMANN_VIEW_ALWAYS_INLINE bool literal_false()
{
return literal("false", 5, value_t::boolean, 0);
}
/// skip whitespace (and comments); false on a malformed comment
NLOHMANN_VIEW_ALWAYS_INLINE bool ws()
{
const unsigned char c = cur();
if (NLOHMANN_VIEW_LIKELY(c > ' ' && (!Comments || c != '/')))
{
return true; // no whitespace: the common case in minified input
}
return ws_slow();
}
NLOHMANN_VIEW_ALWAYS_INLINE bool ws_slow()
{
for (;;)
{
if (cur() == ' ' && (Sentinel || e - p >= 2) && p[1] > ' ' && (!Comments || p[1] != '/'))
{
++p; // single space, e.g. after ':' or ','
return true;
}
if (cur() == '\n' || cur() == '\r')
{
// (a branch, not an add of the comparison: p must not
// wait for the byte after the line break)
if (NLOHMANN_VIEW_UNLIKELY(cur() == '\r') && (Sentinel || e - p >= 2) && p[1] == '\n')
{
p += 2;
}
else
{
++p;
}
// indentation: two spaces per step, fixed offsets (after yyjson)
while (e - p >= 32)
{
#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(load16(p + (std::ptrdiff_t{2} * (i))) == 0x2020)) {} else { p += std::ptrdiff_t{2} * (i); goto indent_done; }
NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP)
#undef NLOHMANN_VIEW_STEP
p += 32;
}
indent_done:
;
}
for (unsigned char c = cur(); c == ' ' || c == '\n' || c == '\r' || c == '\t'; c = cur())
{
++p;
}
if (enabled(Comments) && cur() == '/')
{
const unsigned char* const q = cold.comment(p);
if (q == nullptr)
{
return false;
}
p = q;
continue;
}
return true;
}
}
/// append a node: (kind, flags, extra, off) and the second word (len, or
/// an integer's value); two stores on little-endian targets
NLOHMANN_VIEW_ALWAYS_INLINE node* emit(value_t k, std::uint8_t flags, std::uint16_t extra, std::size_t off, std::uint64_t second)
{
if (NLOHMANN_VIEW_UNLIKELY(out == cap))
{
const auto n = static_cast<std::size_t>(out - base);
base = cold.grow(n, p);
out = base + n;
cap = base + cold.doc.tape_cap;
}
node* n = out++;
#if NLOHMANN_VIEW_LITTLE_ENDIAN
const std::uint64_t first = static_cast<std::uint64_t>(k) | (static_cast<std::uint64_t>(flags) << 8)
| (static_cast<std::uint64_t>(extra) << 16) | (static_cast<std::uint64_t>(off) << 32);
std::memcpy(reinterpret_cast<unsigned char*>(n), &first, 8);
std::memcpy(reinterpret_cast<unsigned char*>(n) + 8, &second, 8);
#else
n->kind = static_cast<std::uint8_t>(k);
n->flags = flags;
n->extra = extra;
n->off = static_cast<std::uint32_t>(off);
set_integer_bits(*n, second);
#endif
return n;
}
NLOHMANN_VIEW_ALWAYS_INLINE void open(value_t k)
{
const auto idx = static_cast<std::uint32_t>(emit(k, 0, 0, static_cast<std::size_t>(p - b), 0) - base);
if (depth != 0)
{
const frame f = {cur_idx, cur_count, cur_is_object};
if (NLOHMANN_VIEW_LIKELY(depth <= 64))
{
cold.shallow[depth - 1] = f;
}
else
{
cold.deep.push_back(f);
}
}
++depth;
cur_idx = idx;
cur_count = 0;
cur_is_object = k == value_t::object;
}
NLOHMANN_VIEW_ALWAYS_INLINE void close()
{
node& n = base[cur_idx];
n.len = cur_count;
n.next = static_cast<std::uint32_t>(out - base) - cur_idx;
if (--depth != 0)
{
frame f{};
if (NLOHMANN_VIEW_LIKELY(depth <= 64))
{
f = cold.shallow[depth - 1];
}
else
{
f = cold.deep.back();
cold.deep.pop_back();
}
cur_idx = f.idx;
cur_count = f.count;
cur_is_object = f.is_object;
}
}
NLOHMANN_VIEW_ALWAYS_INLINE bool literal(const char* text, std::size_t n, value_t k, std::uint8_t flags)
{
if (NLOHMANN_VIEW_UNLIKELY(e - p < static_cast<std::ptrdiff_t>(n) || std::memcmp(p, text, n) != 0))
{
return fail(error_code::invalid_literal);
}
emit(k, flags, 0, static_cast<std::size_t>(p - b), n);
p += n;
return true;
}
/// a number at p; the sign is known from the dispatch (so that p does
/// not have to wait for the first byte)
template<bool negative>
NLOHMANN_VIEW_ALWAYS_INLINE bool number()
{
const unsigned char* const s = p;
if (negative)
{
++p;
}
const unsigned char* const int_start = p;
if (p != e && *p == '0')
{
++p;
}
else if (NLOHMANN_VIEW_LIKELY(p != e && *p >= '1' && *p <= '9'))
{
p = skip_digits(p + 1, e);
}
else
{
return fail(error_code::number_after_minus);
}
const auto int_digits = static_cast<std::size_t>(p - int_start);
std::size_t frac_digits = 0;
bool is_float = false;
if (p != e && *p == '.')
{
++p;
const unsigned char* const f0 = p;
p = skip_digits(p, e);
if (NLOHMANN_VIEW_UNLIKELY(p == f0))
{
return fail(error_code::number_after_dot);
}
frac_digits = static_cast<std::size_t>(p - f0);
is_float = true;
}
std::int64_t exponent = 0;
if (p != e && (*p | 0x20) == 'e')
{
++p;
bool exp_negative = false;
if (p != e && (*p == '+' || *p == '-'))
{
exp_negative = *p == '-';
++p;
}
if (NLOHMANN_VIEW_UNLIKELY(p == e || !is_digit(*p)))
{
return fail(error_code::number_after_exponent);
}
while (p != e && is_digit(*p))
{
if (exponent < 100000)
{
exponent = (exponent * 10) + (*p - '0');
}
++p;
}
if (exp_negative)
{
exponent = -exponent;
}
is_float = true;
}
value_t kind = value_t::number_float;
if (!is_float)
{
kind = negative ? value_t::number_integer : value_t::number_unsigned;
}
if (!is_float)
{
// integers that do not fit become floats, as in parse()
if (NLOHMANN_VIEW_UNLIKELY(int_digits >= 19))
{
if (negative)
{
if (int_digits > 19 || (int_digits == 19 && digits_exceed(int_start, "9223372036854775808", 19)))
{
kind = value_t::number_float;
}
}
else if (int_digits > 20 || (int_digits == 20 && digits_exceed(int_start, "18446744073709551615", 20)))
{
kind = value_t::number_float;
}
}
}
// parse() rejects floats that overflow; only numbers whose magnitude
// could reach the largest FloatType (1e308 for double, 1e38 for
// float) need the conversion
if (NLOHMANN_VIEW_UNLIKELY(static_cast<std::int64_t>(int_digits) + exponent > std::numeric_limits<FloatType>::max_exponent10 - 8 && kind == value_t::number_float))
{
if (builder::float_overflows(s, p))
{
p = s;
return fail(error_code::number_overflow);
}
}
const auto layout = static_cast<std::uint16_t>((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8));
auto second = static_cast<std::uint64_t>(p - s);
if (kind != value_t::number_float)
{
// integers are converted now, while their digits are in cache
const std::uint64_t m = int_digits <= 19 ? parse_upto19(int_start, static_cast<unsigned>(int_digits), e)
: (parse_upto19(int_start, 19, e) * 10) + static_cast<std::uint64_t>(int_start[19] - '0');
second = negative ? 0 - m : m;
}
emit(kind, 0, layout, static_cast<std::size_t>(s - b), second);
return true;
}
/// a string at p: a value (Value) or a key
template<bool Value>
NLOHMANN_VIEW_ALWAYS_INLINE bool string()
{
++p; // opening quote
const unsigned char* const s = p;
p = scan_string_run<Value>(p, e);
if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"'))
{
emit(value_t::string, 0, 0, static_cast<std::size_t>(s - b), static_cast<std::uint64_t>(p - s));
++p;
return true;
}
const decoded r = cold.slow_string(s, p);
if (r.p == nullptr)
{
return false;
}
p = r.p;
emit(value_t::string, node_flags::escaped, 0, r.start, r.len);
return true;
}
};
};
/// run the builder with compile-time options
template<typename FloatType, bool NulIsEnd, bool Comments, bool TrailingCommas>
inline bool build_with(document_data& d, const char* src, std::size_t size, bool sentinel, parse_failure& failure)
{
if (sentinel)
{
builder<FloatType, Comments, TrailingCommas, NulIsEnd, true> bld(d, src, size);
const bool ok = bld.run();
failure = bld.failure();
return ok;
}
builder<FloatType, Comments, TrailingCommas, NulIsEnd, false> bld(d, src, size);
const bool ok = bld.run();
failure = bld.failure();
return ok;
}
/// sentinel: src[size] is readable and 0 (e.g. std::string); FloatType: the
/// number_float_t of the document
template<typename FloatType, bool NulIsEnd>
inline bool build(document_data& d, const char* src, std::size_t size, bool comments, bool trailing_commas, bool sentinel, parse_failure& failure)
{
if (comments)
{
return trailing_commas ? build_with<FloatType, NulIsEnd, true, true>(d, src, size, sentinel, failure)
: build_with<FloatType, NulIsEnd, true, false>(d, src, size, sentinel, failure);
}
return trailing_commas ? build_with<FloatType, NulIsEnd, false, true>(d, src, size, sentinel, failure)
: build_with<FloatType, NulIsEnd, false, false>(d, src, size, sentinel, failure);
}
} // namespace view
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END