diff --git a/BUILD.bazel b/BUILD.bazel
index 1c9b27269..021b8a11e 100644
--- a/BUILD.bazel
+++ b/BUILD.bazel
@@ -78,6 +78,7 @@ cc_library(
"include/nlohmann/detail/view/materialize.hpp",
"include/nlohmann/detail/view/node.hpp",
"include/nlohmann/detail/view/number.hpp",
+ "include/nlohmann/detail/view/object_index.hpp",
"include/nlohmann/detail/view/pointer.hpp",
"include/nlohmann/detail/view/scan.hpp",
"include/nlohmann/detail/view/serializer.hpp",
diff --git a/docs/mkdocs/docs/api/basic_json_view/at.md b/docs/mkdocs/docs/api/basic_json_view/at.md
index d1b1732bb..589e42b7b 100644
--- a/docs/mkdocs/docs/api/basic_json_view/at.md
+++ b/docs/mkdocs/docs/api/basic_json_view/at.md
@@ -74,6 +74,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
another, in document order, stopping at the first match. Each comparison first checks the key's length --
already known from the index, without reading the key bytes -- before comparing its content.
+ Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
+ on average.
2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the
index (unlike `BasicJsonType`'s array, which is random-access).
3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
diff --git a/docs/mkdocs/docs/api/basic_json_view/contains.md b/docs/mkdocs/docs/api/basic_json_view/contains.md
index bbd791887..a48395854 100644
--- a/docs/mkdocs/docs/api/basic_json_view/contains.md
+++ b/docs/mkdocs/docs/api/basic_json_view/contains.md
@@ -35,6 +35,8 @@ No-throw guarantee: this function never throws exceptions.
1. Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
known from the index, without reading the key bytes -- before comparing its content.
+ Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
+ on average.
2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
that level or the index into the array -- as for [`operator[]`](operator[].md#complexity) and
[`at`](at.md#complexity) with a JSON pointer.
diff --git a/docs/mkdocs/docs/api/basic_json_view/count.md b/docs/mkdocs/docs/api/basic_json_view/count.md
index 6a599477e..9cdfe3dee 100644
--- a/docs/mkdocs/docs/api/basic_json_view/count.md
+++ b/docs/mkdocs/docs/api/basic_json_view/count.md
@@ -26,6 +26,8 @@ No-throw guarantee: this function never throws exceptions.
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
known from the index, without reading the key bytes -- before comparing its content.
+Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on
+average.
## Notes
diff --git a/docs/mkdocs/docs/api/basic_json_view/find.md b/docs/mkdocs/docs/api/basic_json_view/find.md
index f7ad92e9f..04ee0476b 100644
--- a/docs/mkdocs/docs/api/basic_json_view/find.md
+++ b/docs/mkdocs/docs/api/basic_json_view/find.md
@@ -28,6 +28,8 @@ No-throw guarantee: this function never throws exceptions.
Linear in the number of members: as for [`ordered_json`](../ordered_json.md), members are compared one after
another, in document order, stopping at the first match. Each comparison first checks the key's length -- already
known from the index, without reading the key bytes -- before comparing its content.
+Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time on
+average.
## Notes
diff --git a/docs/mkdocs/docs/api/basic_json_view/operator[].md b/docs/mkdocs/docs/api/basic_json_view/operator[].md
index 2361bf77c..835800a0a 100644
--- a/docs/mkdocs/docs/api/basic_json_view/operator[].md
+++ b/docs/mkdocs/docs/api/basic_json_view/operator[].md
@@ -76,6 +76,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
another, in document order, stopping at the first match. Each comparison first checks the key's length --
already known from the index, without reading the key bytes -- before comparing its content, so a key of a
different length than `key` is rejected without touching the source text.
+ Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
+ on average.
2. Linear in `idx`: elements are skipped one at a time from the first one, since they are not a fixed size in the
index (unlike `BasicJsonType`'s array, which is random-access).
3. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
diff --git a/docs/mkdocs/docs/api/basic_json_view/value.md b/docs/mkdocs/docs/api/basic_json_view/value.md
index ff63e32f5..06d5372e5 100644
--- a/docs/mkdocs/docs/api/basic_json_view/value.md
+++ b/docs/mkdocs/docs/api/basic_json_view/value.md
@@ -70,6 +70,8 @@ None of these exceptions carry a [`JSON_DIAGNOSTICS`](../macros/json_diagnostics
1. Linear in the number of members: as for [`operator[]`](operator[].md#complexity), members are compared one after
another, in document order, stopping at the first match. Plus the complexity of converting the found member to
`T` (see [`get`](get.md)).
+ Objects with 128 or more members get a hash index while parsing, so that a lookup in them takes constant time
+ on average.
2. Linear in the number of reference tokens of `ptr` and, for each token, in the number of members of the object at
that level or the index into the array -- as for the [`operator[]`](operator[].md#complexity) and
[`at`](at.md#complexity) overloads that take a JSON pointer. Plus the complexity of converting the resolved value
diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp
index 5486cd4d3..cd8088a5a 100644
--- a/include/nlohmann/detail/view/builder.hpp
+++ b/include/nlohmann/detail/view/builder.hpp
@@ -116,6 +116,13 @@ class builder
frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open
std::vector deep{};
+ /// remember an object to index after parsing (out of line, so that the
+ /// parse loop only has a call for it)
+ NLOHMANN_VIEW_NOINLINE void note_large_object(std::uint32_t idx)
+ {
+ doc.large_objects.push_back(idx);
+ }
+
NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept
{
m_failure.code = c;
@@ -628,19 +635,27 @@ obj_next:
if (enabled(TrailingCommas) && cur() == '}')
{
++p;
- goto close_container;
+ goto close_object;
}
goto obj_key;
}
if (cur() == '}')
{
++p;
- goto close_container;
+ goto close_object;
}
return fail(error_code::expected_object_end);
#undef NLOHMANN_VIEW_VALUE
+close_object:
+ // a large object gets a hash index (objects only, so that closing
+ // an array pays nothing for this)
+ if (NLOHMANN_VIEW_UNLIKELY(cur_count >= document_data::index_min_members))
+ {
+ cold.note_large_object(cur_idx);
+ }
+
close_container:
close();
if (NLOHMANN_VIEW_UNLIKELY(depth == 0))
diff --git a/include/nlohmann/detail/view/document_data.hpp b/include/nlohmann/detail/view/document_data.hpp
index d74383a85..d56c1a119 100644
--- a/include/nlohmann/detail/view/document_data.hpp
+++ b/include/nlohmann/detail/view/document_data.hpp
@@ -10,9 +10,11 @@
#include // array
#include // size_t
+#include // uint32_t
#include // memcpy
#include // operator new, placement new
#include // string
+#include // vector
#include
#include
@@ -37,6 +39,17 @@ struct document_data
std::size_t inline_cap = 0;
std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init)
std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init)
+
+ // hash indexes of large objects (see object_index.hpp)
+ static constexpr std::uint32_t index_min_members = 128;
+ struct object_index
+ {
+ std::size_t start; ///< first slot in index_slots
+ std::uint32_t mask; ///< slot count - 1 (a power of two minus one)
+ };
+ std::vector indexes{}; // NOLINT(readability-redundant-member-init)
+ std::vector index_slots{}; // NOLINT(readability-redundant-member-init)
+ std::vector large_objects{}; ///< positions of the objects to index (noted while parsing) // NOLINT(readability-redundant-member-init)
std::array base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage)
bool discarded = true;
diff --git a/include/nlohmann/detail/view/lookup.hpp b/include/nlohmann/detail/view/lookup.hpp
index 71338be32..5d50c0667 100644
--- a/include/nlohmann/detail/view/lookup.hpp
+++ b/include/nlohmann/detail/view/lookup.hpp
@@ -16,6 +16,7 @@
#include
#include
#include
+#include
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -85,6 +86,10 @@ class short_key
/// nullptr; most keys are rejected by their length, from the index alone
inline const node* find_member(const document_data& d, const node* object, const char* key, std::size_t n) noexcept
{
+ if (NLOHMANN_VIEW_UNLIKELY(object->extra != 0))
+ {
+ return find_indexed(d, object, key, n); // a large object
+ }
const node* const end = document_data::child_end(object);
const auto* const k = reinterpret_cast(key); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
if (NLOHMANN_VIEW_LIKELY(n <= 16))
diff --git a/include/nlohmann/detail/view/object_index.hpp b/include/nlohmann/detail/view/object_index.hpp
new file mode 100644
index 000000000..24c5d9ece
--- /dev/null
+++ b/include/nlohmann/detail/view/object_index.hpp
@@ -0,0 +1,129 @@
+// __ _____ _____ _____
+// __| | __| | | | JSON for Modern C++
+// | | |__ | | | | | | version 3.12.0
+// |_____|_____|_____|_|___| https://github.com/nlohmann/json
+//
+// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann
+// SPDX-License-Identifier: MIT
+
+#pragma once
+
+#include // size_t
+#include // uint32_t, uint64_t
+#include // memcmp
+
+#include
+#include
+#include
+#include
+
+// Hash indexes of large objects, so that a lookup does not compare thousands
+// of keys (as Boost.JSON switches from a linear search to a hash table for
+// large objects). An object with document_data::index_min_members members or
+// more gets an open-addressing table after parsing; its node stores the
+// number of the table (1-based) in `extra`. A slot holds the offset of a key
+// node from its object node (0: empty). Of duplicate keys, the first is kept,
+// as for the linear search.
+
+NLOHMANN_JSON_NAMESPACE_BEGIN
+namespace detail
+{
+namespace view
+{
+
+/// hash of a key: its bytes, eight at a time, in a fixed byte order
+inline std::uint64_t key_hash(const char* s, std::size_t n) noexcept
+{
+ std::uint64_t h = 0x9E3779B97F4A7C15u * (n + 1);
+ const auto* p = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
+ while (n >= 8)
+ {
+ h = (h ^ read_eight_bytes(p)) * 0xBF58476D1CE4E5B9u;
+ h ^= h >> 29u;
+ p += 8;
+ n -= 8;
+ }
+ std::uint64_t w = 0;
+ for (std::size_t i = 0; i < n; ++i)
+ {
+ w |= static_cast(p[i]) << (8u * i);
+ }
+ h = (h ^ w) * 0x94D049BB133111EBu;
+ return h ^ (h >> 31u);
+}
+
+/// build the table of a large object
+inline void build_object_index(document_data& d, node* obj)
+{
+ if (d.indexes.size() >= 0xFFFFu)
+ {
+ return; // LCOV_EXCL_LINE (the number must fit `extra`; more large objects are searched linearly)
+ }
+ std::size_t cap = 16;
+ while (cap < 2 * static_cast(obj->len))
+ {
+ cap *= 2;
+ }
+ const std::size_t start = d.index_slots.size();
+ d.index_slots.resize(start + cap, 0);
+ std::uint32_t* const slots = d.index_slots.data() + start;
+ const std::size_t mask = cap - 1;
+ for (const node* k = document_data::first_child(obj), *end = document_data::child_end(obj); k != end; k = document_data::after(k + 1))
+ {
+ const char* const key = d.str(*k);
+ std::size_t i = static_cast(key_hash(key, k->len)) & mask;
+ bool duplicate = false;
+ while (slots[i] != 0)
+ {
+ const node* const other = obj + slots[i];
+ if (other->len == k->len && (k->len == 0 || std::memcmp(d.str(*other), key, k->len) == 0))
+ {
+ duplicate = true; // keep the first
+ break;
+ }
+ i = (i + 1) & mask;
+ }
+ if (!duplicate)
+ {
+ slots[i] = static_cast(k - obj);
+ }
+ }
+ d.indexes.push_back(document_data::object_index{start, static_cast(mask)});
+ obj->extra = static_cast(d.indexes.size());
+}
+
+/// build the tables of the large objects the parser noted
+inline void build_object_indexes(document_data& d)
+{
+ for (const std::uint32_t i : d.large_objects)
+ {
+ build_object_index(d, d.tape + i);
+ }
+}
+
+/// the key node of the first member with this key of an indexed object, or
+/// nullptr
+inline const node* find_indexed(const document_data& d, const node* obj, const char* key, std::size_t n) noexcept
+{
+ const document_data::object_index& ix = d.indexes[obj->extra - 1u];
+ const std::uint32_t* const slots = d.index_slots.data() + ix.start;
+ std::size_t i = static_cast(key_hash(key, n)) & ix.mask;
+ for (;;)
+ {
+ const std::uint32_t s = slots[i];
+ if (s == 0)
+ {
+ return nullptr;
+ }
+ const node* const k = obj + s;
+ if (k->len == n && (n == 0 || std::memcmp(d.str(*k), key, n) == 0))
+ {
+ return k;
+ }
+ i = (i + 1) & ix.mask;
+ }
+}
+
+} // namespace view
+} // namespace detail
+NLOHMANN_JSON_NAMESPACE_END
diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp
index 5ced50c55..7aec0cb2f 100644
--- a/include/nlohmann/json_view.hpp
+++ b/include/nlohmann/json_view.hpp
@@ -25,6 +25,7 @@
#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_
#include // size_t
+#include // uint32_t
#include // memcpy, strlen
#include // distance, input_iterator_tag, iterator_traits
#include