From 82cabbdc6b172d9eb7e568bf1f4e7229fbdce0ee Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Mon, 28 Sep 2026 23:00:46 +0200 Subject: [PATCH] Add dump() to json_view basic_json_view::dump(indent, indent_char, ensure_ascii, number_format) writes the text of a value as ordered_json::parse(text).dump() writes it for the same arguments: members in document order (all of them, should a key occur more than once), strings escaped by the same rules and with the library's scanning kernels, floats with the library's conversion, and integers copied from the source, where they are canonical except "-0". With number_format::source, numbers are copied as they appear in the source ("1.50", "1E2", "-0", all digits of long integers). operator<< takes the indentation from the stream width, as for basic_json. The writer (detail/view/serializer.hpp) writes through a raw pointer into a string sized from the source extent of the value, and walks the index iteratively, so the nesting depth is limited by memory only. Tests compare the output of 2,000 generated documents with ordered_json::dump() for several indentations and ensure_ascii, strings with every kind of escape, numbers (5,000 random doubles, float as number_float_t), duplicate keys, 100,000 levels of nesting, and streams. ViewDump joins the benchmarks. Signed-off-by: Niels Lohmann --- BUILD.bazel | 1 + include/nlohmann/detail/view/serializer.hpp | 404 ++++++++++++++++ include/nlohmann/json_view.hpp | 73 +++ single_include/nlohmann/json_view.hpp | 482 ++++++++++++++++++++ tests/benchmarks/README.md | 1 + tests/benchmarks/src/benchmarks_view.cpp | 34 ++ tests/src/unit-json_view.cpp | 104 +++++ 7 files changed, 1099 insertions(+) create mode 100644 include/nlohmann/detail/view/serializer.hpp diff --git a/BUILD.bazel b/BUILD.bazel index 33c4927a9..20813cff1 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -79,6 +79,7 @@ cc_library( "include/nlohmann/detail/view/number.hpp", "include/nlohmann/detail/view/pointer.hpp", "include/nlohmann/detail/view/scan.hpp", + "include/nlohmann/detail/view/serializer.hpp", "include/nlohmann/detail/view/string_ref.hpp", "include/nlohmann/detail/view/value.hpp", "include/nlohmann/json.hpp", diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp new file mode 100644 index 000000000..ce9bb8e8f --- /dev/null +++ b/include/nlohmann/detail/view/serializer.hpp @@ -0,0 +1,404 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // max +#include // array +#include // isfinite +#include // size_t +#include // uint8_t, uint32_t +#include // memcpy, memset +#include // numeric_limits +#include // integral_constant +#include // vector + +#include +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// append-only output buffer: writes through a raw pointer into a string that +/// is resized ahead, and trimmed by finish() +template +class output_buffer +{ + public: + output_buffer(StringType& out, std::size_t estimate) + : m_out(out) + { + m_out.resize((std::max)(estimate, static_cast(64))); + m_pos = &m_out[0]; + m_end = m_pos + m_out.size(); + } + + void finish() + { + m_out.resize(static_cast(m_pos - m_out.data())); + } + + NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n) + { + if (NLOHMANN_VIEW_UNLIKELY(static_cast(m_end - m_pos) < n)) + { + grow(n); + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE void put(char c) + { + reserve(1); + *m_pos++ = c; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void put(const char* s, std::size_t n) + { + reserve(n); + std::memcpy(m_pos, s, n); + m_pos += n; + } + + void put_repeated(char c, std::size_t n) + { + reserve(n); + std::memset(m_pos, c, n); + m_pos += n; + } + + private: + NLOHMANN_VIEW_NOINLINE void grow(std::size_t n) + { + const std::size_t used = static_cast(m_pos - m_out.data()); + m_out.resize((std::max)(m_out.size() * 2, used + n + 256)); + m_pos = &m_out[0] + used; + m_end = &m_out[0] + m_out.size(); + } + + StringType& m_out; + char* m_pos = nullptr; + char* m_end = nullptr; +}; + +/// how the view's dump() writes a value +struct dump_style +{ + bool pretty = false; ///< indent >= 0 + std::size_t indent = 0; ///< characters per level + char indent_char = ' '; + bool ensure_ascii = false; + bool source_numbers = false; ///< copy number tokens from the source +}; + +/*! +@brief write a view's subtree as basic_json::dump() writes the value + +The output of a subtree equals ordered_json::parse(text).dump() of it for +the same arguments (members in document order): strings are escaped by the +same rules, with the library's scanning kernels; floats are written with +the library's conversion; integers are copied from the source, where they +are canonical (except "-0", which parse() reads as 0). The walk is +iterative, so the nesting depth is limited by memory only. +*/ +template +class view_serializer +{ + using string_t = typename BasicJsonType::string_t; + using number_float_t = typename BasicJsonType::number_float_t; + + public: + view_serializer(const document_data& d, string_t& out, std::size_t estimate, const dump_style& style) + : m_doc(d), m_out(out, estimate), m_style(style) + {} + + void dump(const node* root) + { + struct frame + { + const node* pos; ///< next element, or key of the next member + const node* end; + bool object; + bool first; ///< nothing written yet + }; + std::vector stack; + const node* n = root; + for (;;) + { + // write the value at n + if (is_container(*n)) + { + const bool object = n->kind == static_cast(value_t::object); + if (n->len == 0) + { + m_out.put(object ? "{}" : "[]", 2); + } + else + { + m_out.put(object ? '{' : '['); + stack.push_back(frame{document_data::first_child(n), document_data::child_end(n), object, true}); + } + } + else + { + write_scalar(*n); + } + + // go to the next value: close finished containers, then separate + for (;;) + { + if (stack.empty()) + { + m_out.finish(); + return; + } + frame& f = stack.back(); + if (f.pos == f.end) + { + const bool object = f.object; + stack.pop_back(); + newline(stack.size()); + m_out.put(object ? '}' : ']'); + continue; + } + if (!f.first) + { + m_out.put(','); + } + f.first = false; + newline(stack.size()); + if (f.object) + { + write_string(*f.pos); + if (m_style.pretty) + { + m_out.put(": ", 2); + } + else + { + m_out.put(':'); + } + n = f.pos + 1; + } + else + { + n = f.pos; + } + f.pos = document_data::after(n); + break; + } + } + } + + private: + void newline(std::size_t level) + { + if (m_style.pretty) + { + m_out.put('\n'); + m_out.put_repeated(m_style.indent_char, level * m_style.indent); + } + } + + void write_scalar(const node& n) + { + switch (static_cast(n.kind)) + { + case value_t::null: + m_out.put("null", 4); + break; + case value_t::boolean: + if ((n.flags & node_flags::is_true) != 0) + { + m_out.put("true", 4); + } + else + { + m_out.put("false", 5); + } + break; + case value_t::string: + write_string(n); + break; + case value_t::number_integer: + case value_t::number_unsigned: + { + const char* const token = m_doc.str(n); + const std::uint32_t len = number_length(n); + if (!m_style.source_numbers && len == 2 && token[0] == '-' && token[1] == '0') + { + m_out.put('0'); // parse() reads -0 as the integer 0 + } + else + { + m_out.put(token, len); + } + break; + } + case value_t::number_float: + if (m_style.source_numbers) + { + m_out.put(m_doc.str(n), n.len); + } + else + { + write_float(float_value(m_doc.str(n), n)); + } + break; + case value_t::object: + case value_t::array: + case value_t::binary: + case value_t::discarded: + default: + break; + } + } + + /// as serializer::dump_float() + void write_float(number_float_t x) + { + if (!std::isfinite(x)) + { + m_out.put("null", 4); + return; + } + write_float(x, std::integral_constant < bool, + (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 24 && std::numeric_limits::max_exponent == 128) + || (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53 && std::numeric_limits::max_exponent == 1024) > {}); + } + + void write_float(number_float_t x, std::true_type /*is_ieee_single_or_double*/) + { + std::array buf{}; + const char* const end = ::nlohmann::detail::to_chars(buf.data(), buf.data() + buf.size(), x); + m_out.put(buf.data(), static_cast(end - buf.data())); + } + + void write_float(number_float_t x, std::false_type /*is_ieee_single_or_double*/) + { + // other types (e.g. long double) are rare: the library writes them + const string_t s = BasicJsonType(x).dump(); + m_out.put(s.data(), s.size()); + } + + void write_string(const node& n) + { + const char* const s = m_doc.str(n); + m_out.put('"'); + if ((n.flags & node_flags::escaped) == 0 && !m_style.ensure_ascii) + { + // a string without escape sequences has nothing to escape + m_out.put(s, n.len); + } + else if (m_style.ensure_ascii) + { + write_escaped(reinterpret_cast(s), n.len); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + else + { + write_escaped(reinterpret_cast(s), n.len); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + m_out.put('"'); + } + + /// as serializer::dump_escaped() for valid UTF-8 (the view has no other) + template + void write_escaped(const unsigned char* s, std::size_t n) + { + std::size_t i = 0; + while (i < n) + { + const std::size_t run = EnsureAscii ? (is_ascii_copyable(s[i]) ? find_ascii_copyable_run(s + i, n - i) : 0) + : string_bulk_run(s + i, n - i); + if (run != 0) + { + m_out.put(reinterpret_cast(s + i), run); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + i += run; + continue; + } + std::uint32_t codepoint = s[i]; + std::size_t len = 1; + if (codepoint >= 0xC0) + { + len = codepoint >= 0xF0 ? 4 : (codepoint >= 0xE0 ? 3 : 2); + codepoint &= 0xFFu >> (len + 1); + for (std::size_t k = 1; k < len; ++k) + { + codepoint = (codepoint << 6u) | (s[i + k] & 0x3Fu); + } + } + write_codepoint(codepoint, s + i, len); + i += len; + } + } + + template + void write_codepoint(std::uint32_t codepoint, const unsigned char* bytes, std::size_t len) + { + switch (codepoint) + { + case 0x08: + m_out.put("\\b", 2); + return; + case 0x09: + m_out.put("\\t", 2); + return; + case 0x0A: + m_out.put("\\n", 2); + return; + case 0x0C: + m_out.put("\\f", 2); + return; + case 0x0D: + m_out.put("\\r", 2); + return; + case 0x22: + m_out.put("\\\"", 2); + return; + case 0x5C: + m_out.put("\\\\", 2); + return; + default: + break; + } + if (codepoint <= 0x1F || (EnsureAscii && codepoint >= 0x7F)) + { + if (codepoint <= 0xFFFF) + { + write_u_escape(codepoint); + } + else + { + write_u_escape(0xD7C0u + (codepoint >> 10u)); + write_u_escape(0xDC00u + (codepoint & 0x3FFu)); + } + return; + } + m_out.put(reinterpret_cast(bytes), len); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + void write_u_escape(std::uint32_t u) + { + static constexpr const char* hex = "0123456789abcdef"; + const std::array e = {{'\\', 'u', hex[(u >> 12u) & 0xFu], hex[(u >> 8u) & 0xFu], hex[(u >> 4u) & 0xFu], hex[u & 0xFu]}}; + m_out.put(e.data(), e.size()); + } + + const document_data& m_doc; + output_buffer m_out; + const dump_style m_style; +}; + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp index d9e1ff836..8ef0fc8bb 100644 --- a/include/nlohmann/json_view.hpp +++ b/include/nlohmann/json_view.hpp @@ -29,6 +29,9 @@ #include // distance, input_iterator_tag, iterator_traits #include // map #include // unique_ptr +#ifndef JSON_NO_IO + #include // ostream +#endif #include // string #include // tuple_element, tuple_size #include // decay, enable_if, integral_constant, is_arithmetic, is_base_of, is_integral, is_same, remove_cv, remove_extent @@ -53,6 +56,7 @@ #include #include #include +#include #include #include @@ -547,6 +551,58 @@ class basic_json_view return {m_doc->str(*m_node), detail::view::number_length(*m_node)}; } + /////////////////// + // serialization // + /////////////////// + + /// how dump() writes numbers + enum class number_format + { + /// as basic_json::dump(): integers canonically, floats with the + /// library's shortest round-trip digits ("1.5", "100.0", "1e+100") + shortest, + /// the number text of the source as it is ("1.50", "1E2", "-0", all + /// digits of a long integer) + source, + }; + + /// the text of this value; with number_format::shortest, the output of + /// ordered_json::parse(text).dump() with the same arguments (members in + /// document order, all of them should a key occur more than once) + string_t dump(const int indent = -1, const char indent_char = ' ', const bool ensure_ascii = false, + const number_format numbers = number_format::shortest) const + { + string_t out; + if (m_node == nullptr) + { + out = ""; // as basic_json::dump() of a discarded value + return out; + } + detail::view::dump_style style; + style.pretty = indent >= 0; + style.indent = indent >= 0 ? static_cast(indent) : 0; + style.indent_char = indent_char; + style.ensure_ascii = ensure_ascii; + style.source_numbers = numbers == number_format::source; + // the compact text is about as long as the source text of the value + const std::size_t estimate = source_extent() + (style.pretty ? source_extent() / 2 : 0) + 64; + detail::view::view_serializer(*m_doc, out, estimate, style).dump(m_node); + return out; + } + +#ifndef JSON_NO_IO + /// as operator<< of basic_json: a stream width > 0 is the indentation, + /// the fill character the indentation character + friend std::ostream& operator<<(std::ostream& o, const basic_json_view& v) + { + const bool pretty = o.width() > 0; + const auto indentation = pretty ? o.width() : 0; + o.width(0); + const string_t s = v.dump(pretty ? static_cast(indentation) : -1, o.fill()); + return o.write(s.data(), static_cast(s.size())); + } +#endif + ///////////////// // materialize // ///////////////// @@ -579,6 +635,23 @@ class basic_json_view : m_doc(d), m_node(n) {} + /// the number of source bytes of this value (estimated for values with + /// decoded strings) + std::size_t source_extent() const noexcept + { + const node* const next = document_data::after(m_node); + const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0; + if (!in_source) + { + return m_node->len; + } + if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off) + { + return next->off - m_node->off; + } + return m_doc->size - m_node->off; + } + /// the value of the first member with this key, or a discarded view /// (object required) NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view lookup(string_view_t key) const noexcept diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 8d3c54f87..677430567 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -29,6 +29,9 @@ #include // distance, input_iterator_tag, iterator_traits #include // map #include // unique_ptr +#ifndef JSON_NO_IO + #include // ostream +#endif #include // string #include // tuple_element, tuple_size #include // decay, enable_if, integral_constant, is_arithmetic, is_base_of, is_integral, is_same, remove_cv, remove_extent @@ -2563,6 +2566,416 @@ View resolve_pointer(View cur, const Tokens& tokens, pointer_mode mode) } // namespace detail NLOHMANN_JSON_NAMESPACE_END +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // max +#include // array +#include // isfinite +#include // size_t +#include // uint8_t, uint32_t +#include // memcpy, memset +#include // numeric_limits +#include // integral_constant +#include // vector + +// #include +// #include + +// #include + +// #include + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// append-only output buffer: writes through a raw pointer into a string that +/// is resized ahead, and trimmed by finish() +template +class output_buffer +{ + public: + output_buffer(StringType& out, std::size_t estimate) + : m_out(out) + { + m_out.resize((std::max)(estimate, static_cast(64))); + m_pos = &m_out[0]; + m_end = m_pos + m_out.size(); + } + + void finish() + { + m_out.resize(static_cast(m_pos - m_out.data())); + } + + NLOHMANN_VIEW_ALWAYS_INLINE void reserve(std::size_t n) + { + if (NLOHMANN_VIEW_UNLIKELY(static_cast(m_end - m_pos) < n)) + { + grow(n); + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE void put(char c) + { + reserve(1); + *m_pos++ = c; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void put(const char* s, std::size_t n) + { + reserve(n); + std::memcpy(m_pos, s, n); + m_pos += n; + } + + void put_repeated(char c, std::size_t n) + { + reserve(n); + std::memset(m_pos, c, n); + m_pos += n; + } + + private: + NLOHMANN_VIEW_NOINLINE void grow(std::size_t n) + { + const std::size_t used = static_cast(m_pos - m_out.data()); + m_out.resize((std::max)(m_out.size() * 2, used + n + 256)); + m_pos = &m_out[0] + used; + m_end = &m_out[0] + m_out.size(); + } + + StringType& m_out; + char* m_pos = nullptr; + char* m_end = nullptr; +}; + +/// how the view's dump() writes a value +struct dump_style +{ + bool pretty = false; ///< indent >= 0 + std::size_t indent = 0; ///< characters per level + char indent_char = ' '; + bool ensure_ascii = false; + bool source_numbers = false; ///< copy number tokens from the source +}; + +/*! +@brief write a view's subtree as basic_json::dump() writes the value + +The output of a subtree equals ordered_json::parse(text).dump() of it for +the same arguments (members in document order): strings are escaped by the +same rules, with the library's scanning kernels; floats are written with +the library's conversion; integers are copied from the source, where they +are canonical (except "-0", which parse() reads as 0). The walk is +iterative, so the nesting depth is limited by memory only. +*/ +template +class view_serializer +{ + using string_t = typename BasicJsonType::string_t; + using number_float_t = typename BasicJsonType::number_float_t; + + public: + view_serializer(const document_data& d, string_t& out, std::size_t estimate, const dump_style& style) + : m_doc(d), m_out(out, estimate), m_style(style) + {} + + void dump(const node* root) + { + struct frame + { + const node* pos; ///< next element, or key of the next member + const node* end; + bool object; + bool first; ///< nothing written yet + }; + std::vector stack; + const node* n = root; + for (;;) + { + // write the value at n + if (is_container(*n)) + { + const bool object = n->kind == static_cast(value_t::object); + if (n->len == 0) + { + m_out.put(object ? "{}" : "[]", 2); + } + else + { + m_out.put(object ? '{' : '['); + stack.push_back(frame{document_data::first_child(n), document_data::child_end(n), object, true}); + } + } + else + { + write_scalar(*n); + } + + // go to the next value: close finished containers, then separate + for (;;) + { + if (stack.empty()) + { + m_out.finish(); + return; + } + frame& f = stack.back(); + if (f.pos == f.end) + { + const bool object = f.object; + stack.pop_back(); + newline(stack.size()); + m_out.put(object ? '}' : ']'); + continue; + } + if (!f.first) + { + m_out.put(','); + } + f.first = false; + newline(stack.size()); + if (f.object) + { + write_string(*f.pos); + if (m_style.pretty) + { + m_out.put(": ", 2); + } + else + { + m_out.put(':'); + } + n = f.pos + 1; + } + else + { + n = f.pos; + } + f.pos = document_data::after(n); + break; + } + } + } + + private: + void newline(std::size_t level) + { + if (m_style.pretty) + { + m_out.put('\n'); + m_out.put_repeated(m_style.indent_char, level * m_style.indent); + } + } + + void write_scalar(const node& n) + { + switch (static_cast(n.kind)) + { + case value_t::null: + m_out.put("null", 4); + break; + case value_t::boolean: + if ((n.flags & node_flags::is_true) != 0) + { + m_out.put("true", 4); + } + else + { + m_out.put("false", 5); + } + break; + case value_t::string: + write_string(n); + break; + case value_t::number_integer: + case value_t::number_unsigned: + { + const char* const token = m_doc.str(n); + const std::uint32_t len = number_length(n); + if (!m_style.source_numbers && len == 2 && token[0] == '-' && token[1] == '0') + { + m_out.put('0'); // parse() reads -0 as the integer 0 + } + else + { + m_out.put(token, len); + } + break; + } + case value_t::number_float: + if (m_style.source_numbers) + { + m_out.put(m_doc.str(n), n.len); + } + else + { + write_float(float_value(m_doc.str(n), n)); + } + break; + case value_t::object: + case value_t::array: + case value_t::binary: + case value_t::discarded: + default: + break; + } + } + + /// as serializer::dump_float() + void write_float(number_float_t x) + { + if (!std::isfinite(x)) + { + m_out.put("null", 4); + return; + } + write_float(x, std::integral_constant < bool, + (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 24 && std::numeric_limits::max_exponent == 128) + || (std::numeric_limits::is_iec559 && std::numeric_limits::digits == 53 && std::numeric_limits::max_exponent == 1024) > {}); + } + + void write_float(number_float_t x, std::true_type /*is_ieee_single_or_double*/) + { + std::array buf{}; + const char* const end = ::nlohmann::detail::to_chars(buf.data(), buf.data() + buf.size(), x); + m_out.put(buf.data(), static_cast(end - buf.data())); + } + + void write_float(number_float_t x, std::false_type /*is_ieee_single_or_double*/) + { + // other types (e.g. long double) are rare: the library writes them + const string_t s = BasicJsonType(x).dump(); + m_out.put(s.data(), s.size()); + } + + void write_string(const node& n) + { + const char* const s = m_doc.str(n); + m_out.put('"'); + if ((n.flags & node_flags::escaped) == 0 && !m_style.ensure_ascii) + { + // a string without escape sequences has nothing to escape + m_out.put(s, n.len); + } + else if (m_style.ensure_ascii) + { + write_escaped(reinterpret_cast(s), n.len); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + else + { + write_escaped(reinterpret_cast(s), n.len); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + m_out.put('"'); + } + + /// as serializer::dump_escaped() for valid UTF-8 (the view has no other) + template + void write_escaped(const unsigned char* s, std::size_t n) + { + std::size_t i = 0; + while (i < n) + { + const std::size_t run = EnsureAscii ? (is_ascii_copyable(s[i]) ? find_ascii_copyable_run(s + i, n - i) : 0) + : string_bulk_run(s + i, n - i); + if (run != 0) + { + m_out.put(reinterpret_cast(s + i), run); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + i += run; + continue; + } + std::uint32_t codepoint = s[i]; + std::size_t len = 1; + if (codepoint >= 0xC0) + { + len = codepoint >= 0xF0 ? 4 : (codepoint >= 0xE0 ? 3 : 2); + codepoint &= 0xFFu >> (len + 1); + for (std::size_t k = 1; k < len; ++k) + { + codepoint = (codepoint << 6u) | (s[i + k] & 0x3Fu); + } + } + write_codepoint(codepoint, s + i, len); + i += len; + } + } + + template + void write_codepoint(std::uint32_t codepoint, const unsigned char* bytes, std::size_t len) + { + switch (codepoint) + { + case 0x08: + m_out.put("\\b", 2); + return; + case 0x09: + m_out.put("\\t", 2); + return; + case 0x0A: + m_out.put("\\n", 2); + return; + case 0x0C: + m_out.put("\\f", 2); + return; + case 0x0D: + m_out.put("\\r", 2); + return; + case 0x22: + m_out.put("\\\"", 2); + return; + case 0x5C: + m_out.put("\\\\", 2); + return; + default: + break; + } + if (codepoint <= 0x1F || (EnsureAscii && codepoint >= 0x7F)) + { + if (codepoint <= 0xFFFF) + { + write_u_escape(codepoint); + } + else + { + write_u_escape(0xD7C0u + (codepoint >> 10u)); + write_u_escape(0xDC00u + (codepoint & 0x3FFu)); + } + return; + } + m_out.put(reinterpret_cast(bytes), len); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + void write_u_escape(std::uint32_t u) + { + static constexpr const char* hex = "0123456789abcdef"; + const std::array e = {{'\\', 'u', hex[(u >> 12u) & 0xFu], hex[(u >> 8u) & 0xFu], hex[(u >> 4u) & 0xFu], hex[u & 0xFu]}}; + m_out.put(e.data(), e.size()); + } + + const document_data& m_doc; + output_buffer m_out; + const dump_style m_style; +}; + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include // __ _____ _____ _____ // __| | __| | | | JSON for Modern C++ @@ -3284,6 +3697,58 @@ class basic_json_view return {m_doc->str(*m_node), detail::view::number_length(*m_node)}; } + /////////////////// + // serialization // + /////////////////// + + /// how dump() writes numbers + enum class number_format + { + /// as basic_json::dump(): integers canonically, floats with the + /// library's shortest round-trip digits ("1.5", "100.0", "1e+100") + shortest, + /// the number text of the source as it is ("1.50", "1E2", "-0", all + /// digits of a long integer) + source, + }; + + /// the text of this value; with number_format::shortest, the output of + /// ordered_json::parse(text).dump() with the same arguments (members in + /// document order, all of them should a key occur more than once) + string_t dump(const int indent = -1, const char indent_char = ' ', const bool ensure_ascii = false, + const number_format numbers = number_format::shortest) const + { + string_t out; + if (m_node == nullptr) + { + out = ""; // as basic_json::dump() of a discarded value + return out; + } + detail::view::dump_style style; + style.pretty = indent >= 0; + style.indent = indent >= 0 ? static_cast(indent) : 0; + style.indent_char = indent_char; + style.ensure_ascii = ensure_ascii; + style.source_numbers = numbers == number_format::source; + // the compact text is about as long as the source text of the value + const std::size_t estimate = source_extent() + (style.pretty ? source_extent() / 2 : 0) + 64; + detail::view::view_serializer(*m_doc, out, estimate, style).dump(m_node); + return out; + } + +#ifndef JSON_NO_IO + /// as operator<< of basic_json: a stream width > 0 is the indentation, + /// the fill character the indentation character + friend std::ostream& operator<<(std::ostream& o, const basic_json_view& v) + { + const bool pretty = o.width() > 0; + const auto indentation = pretty ? o.width() : 0; + o.width(0); + const string_t s = v.dump(pretty ? static_cast(indentation) : -1, o.fill()); + return o.write(s.data(), static_cast(s.size())); + } +#endif + ///////////////// // materialize // ///////////////// @@ -3316,6 +3781,23 @@ class basic_json_view : m_doc(d), m_node(n) {} + /// the number of source bytes of this value (estimated for values with + /// decoded strings) + std::size_t source_extent() const noexcept + { + const node* const next = document_data::after(m_node); + const bool in_source = (m_node->flags & detail::view::node_flags::storage) == 0; + if (!in_source) + { + return m_node->len; + } + if (next != m_doc->tape + m_doc->tape_size && (next->flags & detail::view::node_flags::storage) == 0 && next->off >= m_node->off) + { + return next->off - m_node->off; + } + return m_doc->size - m_node->off; + } + /// the value of the first member with this key, or a discarded view /// (object required) NLOHMANN_VIEW_ALWAYS_INLINE basic_json_view lookup(string_view_t key) const noexcept diff --git a/tests/benchmarks/README.md b/tests/benchmarks/README.md index e329a6c8b..7148ecb94 100644 --- a/tests/benchmarks/README.md +++ b/tests/benchmarks/README.md @@ -21,6 +21,7 @@ Micro-benchmarks for parsing, serialization and the binary formats, written with | `ViewParseIndented` | as `ParseIndented`, with a reused `json_document` | | `ViewAccept` | validate with `json_document::accept`; compare with `Accept` | | `ViewMaterialize` | convert a parsed `json_document` into a `json` value | +| `ViewDump` | serialize a parsed `json_document`; compare with `Dump` | The input files are those of [nativejson-benchmark](https://github.com/miloyip/nativejson-benchmark) (`canada`, `citm_catalog`, `twitter`), a large `jeopardy` file, and number-heavy files (`floats`, `signed_ints`, ...). diff --git a/tests/benchmarks/src/benchmarks_view.cpp b/tests/benchmarks/src/benchmarks_view.cpp index d6d04d1d5..475ed54a4 100644 --- a/tests/benchmarks/src/benchmarks_view.cpp +++ b/tests/benchmarks/src/benchmarks_view.cpp @@ -144,3 +144,37 @@ static void ViewMaterialize(benchmark::State& state, const char* filename) state.SetBytesProcessed(state.iterations() * str.size()); } JSON_VIEW_BENCHMARK_FILES(ViewMaterialize); + +////////////////////////////////////////////////////////////////////////////// +// serialize a parsed document (compare with Dump) +////////////////////////////////////////////////////////////////////////////// + +static void ViewDump(benchmark::State& state, const char* filename, int indent) +{ + const std::string str = read_file(filename); + const json_document d = json_document::parse(str); + + while (state.KeepRunning()) + { + std::string output = d.root().dump(indent); + benchmark::DoNotOptimize(output); + } + + state.SetBytesProcessed(state.iterations() * d.root().dump(indent).size()); +} +BENCHMARK_CAPTURE(ViewDump, jeopardy / -, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", -1); +BENCHMARK_CAPTURE(ViewDump, jeopardy / 4, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", 4); +BENCHMARK_CAPTURE(ViewDump, canada / -, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", -1); +BENCHMARK_CAPTURE(ViewDump, canada / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", 4); +BENCHMARK_CAPTURE(ViewDump, citm_catalog / -, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", -1); +BENCHMARK_CAPTURE(ViewDump, citm_catalog / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", 4); +BENCHMARK_CAPTURE(ViewDump, twitter / -, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", -1); +BENCHMARK_CAPTURE(ViewDump, twitter / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", 4); +BENCHMARK_CAPTURE(ViewDump, floats / -, TEST_DATA_DIRECTORY "/regression/floats.json", -1); +BENCHMARK_CAPTURE(ViewDump, floats / 4, TEST_DATA_DIRECTORY "/regression/floats.json", 4); +BENCHMARK_CAPTURE(ViewDump, signed_ints / -, TEST_DATA_DIRECTORY "/regression/signed_ints.json", -1); +BENCHMARK_CAPTURE(ViewDump, signed_ints / 4, TEST_DATA_DIRECTORY "/regression/signed_ints.json", 4); +BENCHMARK_CAPTURE(ViewDump, unsigned_ints / -, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json", -1); +BENCHMARK_CAPTURE(ViewDump, unsigned_ints / 4, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json", 4); +BENCHMARK_CAPTURE(ViewDump, small_signed_ints / -, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json", -1); +BENCHMARK_CAPTURE(ViewDump, small_signed_ints / 4, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json", 4); diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 640310539..9fb9ff411 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -22,6 +22,7 @@ using nlohmann::ordered_json_view; #include #include #include +#include #include #include #include @@ -1081,3 +1082,106 @@ TEST_CASE("json_view JSON pointers") } #endif } + +TEST_CASE("json_view dump") +{ + SECTION("the output of ordered_json::dump()") + { + generator g; + for (int i = 0; i < 2000; ++i) + { + std::string text; + g.value(text, 0); + const ordered_json_document d = ordered_json_document::parse(text); + if (has_duplicate_keys(d.root())) + { + continue; + } + CAPTURE(text); + const ordered_json j = ordered_json::parse(text); + for (const int indent : + { + -1, 0, 2 + }) + { + for (const bool ensure_ascii : + { + false, true + }) + { + CHECK(d.root().dump(indent, i % 2 == 0 ? ' ' : '\t', ensure_ascii) == j.dump(indent, i % 2 == 0 ? ' ' : '\t', ensure_ascii)); + } + } + // also of each element + for (const ordered_json_view e : d.root()) + { + CHECK(e.dump() == e.materialize().dump()); + } + } + } + + SECTION("strings") + { + const std::string text = R"(["plain", "\u0000\u0001\u001f\u007f\u0080é€￿😀", "\"\\\/\b\f\n\r\t", "aéあ😀b", "long text beyond the eight bytes of a word \n with an escape in the middle"])"; + const ordered_json_document d = ordered_json_document::parse(text); + const ordered_json j = ordered_json::parse(text); + CHECK(d.root().dump() == j.dump()); + CHECK(d.root().dump(-1, ' ', true) == j.dump(-1, ' ', true)); + CHECK(d.root().dump(4, ' ', true) == j.dump(4, ' ', true)); + const ordered_json_document keys = ordered_json_document::parse(R"({"é\n": {"\"": [], "": {}}})"); + CHECK(keys.root().dump(2, ' ', true) == ordered_json::parse(R"({"é\n": {"\"": [], "": {}}})").dump(2, ' ', true)); + } + + SECTION("numbers") + { + const std::string text = "[1.50, 1E2, -0, -0.0, 123456789012345678901234567890, 18446744073709551615, -9223372036854775808, 0.1, 1e-7, 5e-324]"; + const json_document d = json_document::parse(text); + CHECK(d.root().dump() == json::parse(text).dump()); + CHECK(d.root().dump() == "[1.5,100.0,0,-0.0,1.2345678901234568e+29,18446744073709551615,-9223372036854775808,0.1,1e-07,5e-324]"); + CHECK(d.root().dump(-1, ' ', false, json_view::number_format::source) == "[1.50,1E2,-0,-0.0,123456789012345678901234567890,18446744073709551615,-9223372036854775808,0.1,1e-7,5e-324]"); + + // random doubles, written as parse() and dump() would + std::mt19937_64 rng(1170); + std::string many = "["; + for (int i = 0; i < 5000; ++i) + { + const std::uint64_t bits = rng(); + double x = 0; + std::memcpy(&x, &bits, sizeof(x)); + if (std::isfinite(x)) + { + many += (many.size() > 1 ? "," : "") + json(x).dump(); + } + } + many += "]"; + CHECK(json_document::parse(many).root().dump() == json::parse(many).dump()); + + using json_float = nlohmann::basic_json; + CHECK(nlohmann::basic_json_document::parse("[0.1, 1.5e10, 3.4028235e38]").root().dump() == json_float::parse("[0.1, 1.5e10, 3.4028235e38]").dump()); + } + + SECTION("members in document order, all of them") + { + const json_document d = json_document::parse(R"({"b": 1, "a": 2, "b": 3})"); + CHECK(d.root().dump() == R"({"b":1,"a":2,"b":3})"); + CHECK(d.root().dump(1) == "{\n \"b\": 1,\n \"a\": 2,\n \"b\": 3\n}"); + } + + SECTION("deep nesting") + { + const std::string deep = std::string(100000, '[') + std::string(100000, ']'); + CHECK(json_document::parse(deep).root().dump() == deep); + } + + SECTION("streams and discarded views") + { + const json_document d = json_document::parse(R"({"a": [1, 2]})"); + std::ostringstream compact; + compact << d.root(); + CHECK(compact.str() == R"({"a":[1,2]})"); + std::ostringstream pretty; + pretty << std::setw(2) << std::setfill('.') << d.root() << d.root()["a"]; + CHECK(pretty.str() == "{\n..\"a\": [\n....1,\n....2\n..]\n}[1,2]"); + CHECK(json_view().dump() == json(json::value_t::discarded).dump()); + } +}