diff --git a/BUILD.bazel b/BUILD.bazel index 1c9b27269..021b8a11e 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -78,6 +78,7 @@ cc_library( "include/nlohmann/detail/view/materialize.hpp", "include/nlohmann/detail/view/node.hpp", "include/nlohmann/detail/view/number.hpp", + "include/nlohmann/detail/view/object_index.hpp", "include/nlohmann/detail/view/pointer.hpp", "include/nlohmann/detail/view/scan.hpp", "include/nlohmann/detail/view/serializer.hpp", diff --git a/docs/mkdocs/docs/api/basic_json_view/at.md b/docs/mkdocs/docs/api/basic_json_view/at.md index d1b1732bb..589e42b7b 100644 --- a/docs/mkdocs/docs/api/basic_json_view/at.md +++ b/docs/mkdocs/docs/api/basic_json_view/at.md @@ -74,6 +74,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content. + Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time + on average. 2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the index (unlike `BasicJsonType`'s array, which is random-access). 3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at diff --git a/docs/mkdocs/docs/api/basic_json_view/contains.md b/docs/mkdocs/docs/api/basic_json_view/contains.md index bbd791887..a48395854 100644 --- a/docs/mkdocs/docs/api/basic_json_view/contains.md +++ b/docs/mkdocs/docs/api/basic_json_view/contains.md @@ -35,6 +35,8 @@ No-throw guarantee: this function never throws exceptions. 1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content. + Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time + on average. 2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at that level or the index into the array -- as for [`operator[]`](operator[].md#complexity) and [`at`](at.md#complexity) with a JSON pointer. diff --git a/docs/mkdocs/docs/api/basic_json_view/count.md b/docs/mkdocs/docs/api/basic_json_view/count.md index 6a599477e..9cdfe3dee 100644 --- a/docs/mkdocs/docs/api/basic_json_view/count.md +++ b/docs/mkdocs/docs/api/basic_json_view/count.md @@ -26,6 +26,8 @@ No-throw guarantee: this function never throws exceptions. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content. +Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on +average. ## Notes diff --git a/docs/mkdocs/docs/api/basic_json_view/find.md b/docs/mkdocs/docs/api/basic_json_view/find.md index f7ad92e9f..04ee0476b 100644 --- a/docs/mkdocs/docs/api/basic_json_view/find.md +++ b/docs/mkdocs/docs/api/basic_json_view/find.md @@ -28,6 +28,8 @@ No-throw guarantee: this function never throws exceptions. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content. +Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on +average. ## Notes diff --git a/docs/mkdocs/docs/api/basic_json_view/operator[].md b/docs/mkdocs/docs/api/basic_json_view/operator[].md index 2361bf77c..835800a0a 100644 --- a/docs/mkdocs/docs/api/basic_json_view/operator[].md +++ b/docs/mkdocs/docs/api/basic_json_view/operator[].md @@ -76,6 +76,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics another, in document order, stopping at the first match. Each comparison first checks the key's length -- already known from the index, without reading the key bytes -- before comparing its content, so a key of a different length than `key` is rejected without touching the source text. + Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time + on average. 2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the index (unlike `BasicJsonType`'s array, which is random-access). 3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at diff --git a/docs/mkdocs/docs/api/basic_json_view/value.md b/docs/mkdocs/docs/api/basic_json_view/value.md index ff63e32f5..06d5372e5 100644 --- a/docs/mkdocs/docs/api/basic_json_view/value.md +++ b/docs/mkdocs/docs/api/basic_json_view/value.md @@ -70,6 +70,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics 1. Linear in the number of members: as for [`operator[]`](operator[].md#complexity), members are compared one after another, in document order, stopping at the first match. Plus the complexity of converting the found member to `T` (see [`get`](get.md)). + Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time + on average. 2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at that level or the index into the array -- as for the [`operator[]`](operator[].md#complexity) and [`at`](at.md#complexity) overloads that take a JSON pointer. Plus the complexity of converting the resolved value diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp index 5486cd4d3..cd8088a5a 100644 --- a/include/nlohmann/detail/view/builder.hpp +++ b/include/nlohmann/detail/view/builder.hpp @@ -116,6 +116,13 @@ class builder frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open std::vector deep{}; + /// remember an object to index after parsing (out of line, so that the + /// parse loop only has a call for it) + NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx) + { + doc.large_objects.push_back(idx); + } + NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept { m_failure.code = c; @@ -628,19 +635,27 @@ obj_next: if (enabled(TrailingCommas) && cur() == '}') { ++p; - goto close_container; + goto close_object; } goto obj_key; } if (cur() == '}') { ++p; - goto close_container; + goto close_object; } return fail(error_code::expected_object_end); #undef NLOHMANN_VIEW_VALUE +close_object: + // a large object gets a hash index (objects only, so that closing + // an array pays nothing for this) + if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members)) + { + cold.note_large_object(cur_idx); + } + close_container: close(); if (NLOHMANN_VIEW_UNLIKELY(depth == 0)) diff --git a/include/nlohmann/detail/view/document_data.hpp b/include/nlohmann/detail/view/document_data.hpp index d74383a85..d56c1a119 100644 --- a/include/nlohmann/detail/view/document_data.hpp +++ b/include/nlohmann/detail/view/document_data.hpp @@ -10,9 +10,11 @@ #include // array #include // size_t +#include // uint32_t #include // memcpy #include // operator new, placement new #include // string +#include // vector #include #include @@ -37,6 +39,17 @@ struct document_data std::size_t inline_cap = 0; std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init) std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init) + + // hash indexes of large objects (see object_index.hpp) + static constexpr std::uint32_t index_min_members = 128; + struct object_index + { + std::size_t start; ///< first slot in index_slots + std::uint32_t mask; ///< slot count - 1 (a power of two minus one) + }; + std::vector indexes{}; // NOLINT(readability-redundant-member-init) + std::vector index_slots{}; // NOLINT(readability-redundant-member-init) + std::vector large_objects{}; ///< positions of the objects to index (noted while parsing) // NOLINT(readability-redundant-member-init) std::array base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage) bool discarded = true; diff --git a/include/nlohmann/detail/view/lookup.hpp b/include/nlohmann/detail/view/lookup.hpp index 71338be32..5d50c0667 100644 --- a/include/nlohmann/detail/view/lookup.hpp +++ b/include/nlohmann/detail/view/lookup.hpp @@ -16,6 +16,7 @@ #include #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -85,6 +86,10 @@ class short_key /// nullptr; most keys are rejected by their length, from the index alone inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { + if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0)) + { + return find_indexed(d, object, key, n); // a large object + } const node* const end = document_data::child_end(object); const auto* const k = reinterpret_cast(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) if (NLOHMANN_VIEW_LIKELY(n <= 16)) diff --git a/include/nlohmann/detail/view/object_index.hpp b/include/nlohmann/detail/view/object_index.hpp new file mode 100644 index 000000000..24c5d9ece --- /dev/null +++ b/include/nlohmann/detail/view/object_index.hpp @@ -0,0 +1,129 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t +#include // uint32_t, uint64_t +#include // memcmp + +#include +#include +#include +#include + +// Hash indexes of large objects, so that a lookup does not compare thousands +// of keys (as Boost.JSON switches from a linear search to a hash table for +// large objects). An object with document_data::index_min_members members or +// more gets an open-addressing table after parsing; its node stores the +// number of the table (1-based) in `extra`. A slot holds the offset of a key +// node from its object node (0: empty). Of duplicate keys, the first is kept, +// as for the linear search. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// hash of a key: its bytes, eight at a time, in a fixed byte order +inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept +{ + std::uint64_t h = 0x9E3779B97F4A7C15u * (n + 1); + const auto* p = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + while (n >= 8) + { + h = (h ^ read_eight_bytes(p)) * 0xBF58476D1CE4E5B9u; + h ^= h >> 29u; + p += 8; + n -= 8; + } + std::uint64_t w = 0; + for (std::size_t i = 0; i < n; ++i) + { + w |= static_cast(p[i]) << (8u * i); + } + h = (h ^ w) * 0x94D049BB133111EBu; + return h ^ (h >> 31u); +} + +/// build the table of a large object +inline void build_object_index(document_data& d, node* obj) +{ + if (d.indexes.size() >= 0xFFFFu) + { + return; // LCOV_EXCL_LINE (the number must fit `extra`; more large objects are searched linearly) + } + std::size_t cap = 16; + while (cap < 2 * static_cast(obj->len)) + { + cap *= 2; + } + const std::size_t start = d.index_slots.size(); + d.index_slots.resize(start + cap, 0); + std::uint32_t* const slots = d.index_slots.data() + start; + const std::size_t mask = cap - 1; + for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1)) + { + const char* const key = d.str(*k); + std::size_t i = static_cast(key_hash(key, k->len)) & mask; + bool duplicate = false; + while (slots[i] != 0) + { + const node* const other = obj + slots[i]; + if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) + { + duplicate = true; // keep the first + break; + } + i = (i + 1) & mask; + } + if (!duplicate) + { + slots[i] = static_cast(k - obj); + } + } + d.indexes.push_back(document_data::object_index{start, static_cast(mask)}); + obj->extra = static_cast(d.indexes.size()); +} + +/// build the tables of the large objects the parser noted +inline void build_object_indexes(document_data& d) +{ + for (const std::uint32_t i : d.large_objects) + { + build_object_index(d, d.tape + i); + } +} + +/// the key node of the first member with this key of an indexed object, or +/// nullptr +inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept +{ + const document_data::object_index& ix = d.indexes[obj->extra - 1u]; + const std::uint32_t* const slots = d.index_slots.data() + ix.start; + std::size_t i = static_cast(key_hash(key, n)) & ix.mask; + for (;;) + { + const std::uint32_t s = slots[i]; + if (s == 0) + { + return nullptr; + } + const node* const k = obj + s; + if (k->len == n && (n == 0 || std::memcmp(d.str(*k), key, n) == 0)) + { + return k; + } + i = (i + 1) & ix.mask; + } +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index 5ced50c55..7aec0cb2f 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -25,6 +25,7 @@ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #include // size_t +#include // uint32_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits #include // map @@ -56,6 +57,7 @@ #include #include #include +#include #include #include #include @@ -925,7 +927,9 @@ class basic_json_document } return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node)) + (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0) - + m_data->arena.capacity() + m_data->owned.capacity(); + + m_data->arena.capacity() + m_data->owned.capacity() + + (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t)) + + (m_data->large_objects.capacity() * sizeof(std::uint32_t)); } /// release unused capacity of the index and the decoded strings; like @@ -995,6 +999,9 @@ class basic_json_document d.size = size; d.tape_size = 0; d.arena.clear(); + d.indexes.clear(); + d.index_slots.clear(); + d.large_objects.clear(); d.discarded = true; detail::view::parse_failure failure; bool ok = false; @@ -1010,6 +1017,7 @@ class basic_json_document { d.base[0] = d.src; d.base[1] = d.arena.data(); + detail::view::build_object_indexes(d); d.discarded = false; return; } diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 8c06e0b89..07c2ab5f5 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -25,6 +25,7 @@ #define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ #include // size_t +#include // uint32_t #include // memcpy, strlen #include // distance, input_iterator_tag, iterator_traits #include // map @@ -81,9 +82,11 @@ #include // array #include // size_t +#include // uint32_t #include // memcpy #include // operator new, placement new #include // string +#include // vector // #include // #include @@ -279,6 +282,17 @@ struct document_data std::size_t inline_cap = 0; std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init) std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init) + + // hash indexes of large objects (see object_index.hpp) + static constexpr std::uint32_t index_min_members = 128; + struct object_index + { + std::size_t start; ///< first slot in index_slots + std::uint32_t mask; ///< slot count - 1 (a power of two minus one) + }; + std::vector indexes{}; // NOLINT(readability-redundant-member-init) + std::vector index_slots{}; // NOLINT(readability-redundant-member-init) + std::vector large_objects{}; ///< positions of the objects to index (noted while parsing) // NOLINT(readability-redundant-member-init) std::array base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage) bool discarded = true; @@ -967,6 +981,13 @@ class builder frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open std::vector deep{}; + /// remember an object to index after parsing (out of line, so that the + /// parse loop only has a call for it) + NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx) + { + doc.large_objects.push_back(idx); + } + NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept { m_failure.code = c; @@ -1479,19 +1500,27 @@ obj_next: if (enabled(TrailingCommas) && cur() == '}') { ++p; - goto close_container; + goto close_object; } goto obj_key; } if (cur() == '}') { ++p; - goto close_container; + goto close_object; } return fail(error_code::expected_object_end); #undef NLOHMANN_VIEW_VALUE +close_object: + // a large object gets a hash index (objects only, so that closing + // an array pays nothing for this) + if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members)) + { + cold.note_large_object(cur_idx); + } + close_container: close(); if (NLOHMANN_VIEW_UNLIKELY(depth == 0)) @@ -2687,6 +2716,140 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t +#include // uint32_t, uint64_t +#include // memcmp + +// #include +// #include + +// #include + +// #include + + +// Hash indexes of large objects, so that a lookup does not compare thousands +// of keys (as Boost.JSON switches from a linear search to a hash table for +// large objects). An object with document_data::index_min_members members or +// more gets an open-addressing table after parsing; its node stores the +// number of the table (1-based) in `extra`. A slot holds the offset of a key +// node from its object node (0: empty). Of duplicate keys, the first is kept, +// as for the linear search. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// hash of a key: its bytes, eight at a time, in a fixed byte order +inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept +{ + std::uint64_t h = 0x9E3779B97F4A7C15u * (n + 1); + const auto* p = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + while (n >= 8) + { + h = (h ^ read_eight_bytes(p)) * 0xBF58476D1CE4E5B9u; + h ^= h >> 29u; + p += 8; + n -= 8; + } + std::uint64_t w = 0; + for (std::size_t i = 0; i < n; ++i) + { + w |= static_cast(p[i]) << (8u * i); + } + h = (h ^ w) * 0x94D049BB133111EBu; + return h ^ (h >> 31u); +} + +/// build the table of a large object +inline void build_object_index(document_data& d, node* obj) +{ + if (d.indexes.size() >= 0xFFFFu) + { + return; // LCOV_EXCL_LINE (the number must fit `extra`; more large objects are searched linearly) + } + std::size_t cap = 16; + while (cap < 2 * static_cast(obj->len)) + { + cap *= 2; + } + const std::size_t start = d.index_slots.size(); + d.index_slots.resize(start + cap, 0); + std::uint32_t* const slots = d.index_slots.data() + start; + const std::size_t mask = cap - 1; + for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1)) + { + const char* const key = d.str(*k); + std::size_t i = static_cast(key_hash(key, k->len)) & mask; + bool duplicate = false; + while (slots[i] != 0) + { + const node* const other = obj + slots[i]; + if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0)) + { + duplicate = true; // keep the first + break; + } + i = (i + 1) & mask; + } + if (!duplicate) + { + slots[i] = static_cast(k - obj); + } + } + d.indexes.push_back(document_data::object_index{start, static_cast(mask)}); + obj->extra = static_cast(d.indexes.size()); +} + +/// build the tables of the large objects the parser noted +inline void build_object_indexes(document_data& d) +{ + for (const std::uint32_t i : d.large_objects) + { + build_object_index(d, d.tape + i); + } +} + +/// the key node of the first member with this key of an indexed object, or +/// nullptr +inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept +{ + const document_data::object_index& ix = d.indexes[obj->extra - 1u]; + const std::uint32_t* const slots = d.index_slots.data() + ix.start; + std::size_t i = static_cast(key_hash(key, n)) & ix.mask; + for (;;) + { + const std::uint32_t s = slots[i]; + if (s == 0) + { + return nullptr; + } + const node* const k = obj + s; + if (k->len == n && (n == 0 || std::memcmp(d.str(*k), key, n) == 0)) + { + return k; + } + i = (i + 1) & ix.mask; + } +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -2756,6 +2919,10 @@ class short_key /// nullptr; most keys are rejected by their length, from the index alone inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept { + if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0)) + { + return find_indexed(d, object, key, n); // a large object + } const node* const end = document_data::child_end(object); const auto* const k = reinterpret_cast(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) if (NLOHMANN_VIEW_LIKELY(n <= 16)) @@ -3115,6 +3282,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -4809,7 +4978,9 @@ class basic_json_document } return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node)) + (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0) - + m_data->arena.capacity() + m_data->owned.capacity(); + + m_data->arena.capacity() + m_data->owned.capacity() + + (m_data->indexes.capacity() * sizeof(document_data::object_index)) + (m_data->index_slots.capacity() * sizeof(std::uint32_t)) + + (m_data->large_objects.capacity() * sizeof(std::uint32_t)); } /// release unused capacity of the index and the decoded strings; like @@ -4879,6 +5050,9 @@ class basic_json_document d.size = size; d.tape_size = 0; d.arena.clear(); + d.indexes.clear(); + d.index_slots.clear(); + d.large_objects.clear(); d.discarded = true; detail::view::parse_failure failure; bool ok = false; @@ -4894,6 +5068,7 @@ class basic_json_document { d.base[0] = d.src; d.base[1] = d.arena.data(); + detail::view::build_object_indexes(d); d.discarded = false; return; } diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 6729df7e6..e829c3198 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1277,3 +1277,60 @@ TEST_CASE("json_view comparison") CHECK(a.root() != json_document::parse(other).root()); } } + +TEST_CASE("json_view large objects") +{ + // objects with 128 members or more are looked up with a hash index + for (const std::size_t members : + { + 127u, 128u, 129u, 10000u + }) + { + CAPTURE(members); + std::string text = "{"; + for (std::size_t i = 0; i < members; ++i) + { + text += (i != 0 ? ",\"" : "\"") + std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\\n" : "") + "\":" + std::to_string(i); + } + text += ",\"\":\"empty key\",\"k1\":\"a duplicate of an earlier key\"}"; + const json_document d = json_document::parse(text); + const json_view v = d.root(); + const json j = json::parse(text); + for (std::size_t i = 0; i < members; ++i) + { + const std::string key = std::string(i % 23, 'k') + std::to_string(i) + (i % 7 == 0 ? "\n" : ""); + CHECK(v[key].get() == i); + CHECK(v.contains(key)); + CHECK(v.find(key).key() == key); + CHECK(v.at(key).get() == i); + CHECK(!v.contains(key + "x")); + } + CHECK(v[""].get_string() == "empty key"); + CHECK(v["k1"].get() == 1); // the first of duplicate keys, as for small objects + CHECK(!v.contains("missing")); + CHECK_THROWS_WITH_AS(v.at("missing"), "[json.exception.out_of_range.403] key 'missing' not found", json::out_of_range&); + CHECK(v == j); + CHECK(v.materialize() == j); + } + + SECTION("nested, reused, and in arrays") + { + std::string inner = "{"; + for (int i = 0; i < 300; ++i) + { + inner += (i != 0 ? ",\"m" : "\"m") + std::to_string(i) + "\":" + std::to_string(i); + } + inner += "}"; + const std::string text = "[" + inner + ",{\"x\":" + inner + "}," + inner + "]"; + json_document d = json_document::parse(text); + CHECK(d.root()[0]["m299"].get() == 299); + CHECK(d.root()[1]["x"]["m150"].get() == 150); + CHECK(d.root()[2]["m0"].get() == 0); + const std::size_t with_index = d.memory_usage(); + d.read(std::string("{\"small\": 1}")); + CHECK(d.root()["small"].get() == 1); + d.read(text); + CHECK(d.root()[2]["m7"].get() == 7); + CHECK(d.memory_usage() >= with_index / 2); + } +}