From ebcc734e452d111403d72c929f7a785bf2a12e66 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 29 Sep 2026 03:33:19 +0200 Subject: [PATCH] Write the floats of json_view from their digits dump() writes a float token of at most 15 significant digits from its digits, without converting it to a double and back: two decimals of at most 15 digits are farther apart than the rounding interval of a normal double (the argument behind DBL_DIG), so the token's digits are the shortest ones of its double, which the library's conversion (Zmij) writes. The exponent must keep the value away from subnormals and overflow. Longer tokens are converted from the digits already read. Doubles are written into the output directly instead of through a local buffer. With NEON, the fixed layouts ("12.5", "0.001", "100.0") are put together in vector registers by a table lookup of the digit bytes: the portable layout copies the digits through a buffer at another offset, and a load that spans several recent stores waits until they reach the cache. dump() of float-heavy documents: numbers -69%, marine_ik -62%, mesh.pretty -34%, canada (mostly 16 or 17 digits) -14%. Tests: 20,000 float tokens of 1 to 17 significant digits in every spelling (point, exponent, leading and trailing zeros, sign), from about 1e-320 to 1e300, written as json::dump() writes them. On AArch64 they check the NEON layout; x86 and JSON_VIEW_NO_SIMD use the library's. Other float types, now the only ones on the general path, are tested with non-finite values set by edits (written as null). Signed-off-by: Niels Lohmann --- include/nlohmann/detail/view/serializer.hpp | 150 ++++++++++++++++++- single_include/nlohmann/json_view.hpp | 151 +++++++++++++++++++- tests/src/unit-json_view.cpp | 41 ++++++ tests/src/unit-json_view_edit.cpp | 14 ++ 4 files changed, 352 insertions(+), 4 deletions(-) diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index 42c904b83..f471f1642 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -23,6 +23,7 @@ #include #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -221,6 +222,69 @@ struct dump_style bool source_numbers = false; ///< copy number tokens from the source }; +/*! +@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0, +at most 17 digits; up to 41 bytes are written at first) + +With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in +vector registers: output byte i is byte s + i of the digits (after '0's) +before the point, and byte s + i - 1 after it. The portable code writes the +digits to a buffer and copies them from there at another offset, and a load +that spans several recent stores waits until they reach the cache. +*/ +NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept +{ +#if NLOHMANN_VIEW_NEON + namespace dtoa = ::nlohmann::detail::dtoa_impl; + const std::uint64_t upper = digits / 100000000u; + const std::uint64_t b0 = upper / 100000000u; // one digit + const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u); + const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u); + // leading and trailing zero digits (as dtoa_impl::write_decimal()) + int leading = 7; + if (b0 == 0) + { + leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8); + } + int zeros = 16; + if (b2 != 0) + { + zeros = count_trailing_zeros(b2) / 8; + } + else if (b1 != 0) + { + zeros = 8 + (count_trailing_zeros(b1) / 8); + } + const int k = 24 - leading - zeros; // significant digits + const int n = k + exp + zeros; // position of the point after the first digit + if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15)) + { + static const std::array iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}}; + const int pad = n <= 0 ? 1 - n : 0; + const int len = k + pad; + const int point = n + pad; + // the 24 digit bytes in memory order, then '0's + const std::uint64_t zero_chars = 0x3030303030303030u; + const uint8x16x2_t table = {{ + vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))), + vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0')) + } + }; + const uint8x16_t s = vdupq_n_u8(static_cast(leading - pad)); + const uint8x16_t at_point = vdupq_n_u8(static_cast(point)); + for (std::size_t half = 0; half < 2; ++half) + { + const uint8x16_t i = vld1q_u8(iota.data() + (16 * half)); + // (+ 0xFF is - 1 after the point; indexes past the digits read a '0') + const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31)); + vst1q_u8(reinterpret_cast(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0" + } +#endif + return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp); +} + /*! @brief write a view's subtree as basic_json::dump() writes the value @@ -496,10 +560,15 @@ class view_serializer room(n->len); copy(src + n->off, n->len); } + else if (std::is_same::value) + { + room(64); + w = write_double_at(w, *n); + } else { m_out.set_cursor(w); - write_float(float_value(m_doc, *n)); + write_float_node(*n); w = m_out.cursor(); lim = m_out.limit(); } @@ -666,7 +735,7 @@ class view_serializer } else { - write_float(float_value(m_doc, n)); + write_float_node(n); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) @@ -678,6 +747,83 @@ class view_serializer } } + /// a float node as dump() writes it + void write_float_node(const node& n) + { + write_float_node(n, std::is_same {}); + } + + void write_float_node(const node& n, std::false_type /*other*/) + { + write_float(float_value(m_doc, n)); + } + + void write_float_node(const node& n, std::true_type /*double*/) + { + m_out.reserve(64); + m_out.set_cursor(write_double_at(m_out.cursor(), n)); + } + + /*! + @brief (doubles) the float at n as dump() writes it, at w (64 bytes of room) + + A token of at most 15 significant digits is written from its digits, + without a conversion: two decimals of at most 15 digits are farther + apart than the rounding interval of a (normal) double (the argument + behind DBL_DIG), so the token's digits are the shortest ones of its + double, which the library's conversion writes (Zmij). Other tokens are + converted from the digits already read. + */ + char* write_double_at(char* w, const node& n) + { + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if ((n.flags & node_flags::storage) != node_flags::edited && int_digits + frac_digits <= 19) + { + const auto* const first = reinterpret_cast(m_doc.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const float_significand d = layout_decimal(first, first + n.len, int_digits, frac_digits, reinterpret_cast(m_doc.src + m_doc.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + // (the exponent keeps the value far from subnormals and overflow) + if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290) + { + *w = '-'; + return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast(d.exponent)); + } + return write_double_value_at(w, decimal_to_float(d)); // (without reading the token again) + } + return write_double_value_at(w, static_cast(float_value(m_doc, n))); + } + + /// n bytes of text at w + static char* write_text_at(char* w, const char* text, std::size_t n) noexcept + { + std::memcpy(w, text, n); + return w + n; + } + + /// a double as dump() writes it, at w (64 bytes of room) + static char* write_double_value_at(char* w, double x) + { + if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x))) + { + return write_text_at(w, "null", 4); + } +#if NLOHMANN_VIEW_NEON + std::uint64_t bits = 0; + std::memcpy(&bits, &x, sizeof(bits)); + *w = '-'; + w += bits >> 63u; + bits &= ~(std::uint64_t{1} << 63u); + if (bits == 0) + { + return write_text_at(w, "0.0", 3); + } + const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits); + return write_decimal(w, d.significand, d.exponent); +#else + return ::nlohmann::detail::to_chars(w, w + 64, x); +#endif + } + /// as serializer::dump_float() void write_float(number_float_t x) { diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 6ec509c0c..faa3ac437 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -5323,6 +5323,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -5521,6 +5523,69 @@ struct dump_style bool source_numbers = false; ///< copy number tokens from the source }; +/*! +@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0, +at most 17 digits; up to 41 bytes are written at first) + +With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in +vector registers: output byte i is byte s + i of the digits (after '0's) +before the point, and byte s + i - 1 after it. The portable code writes the +digits to a buffer and copies them from there at another offset, and a load +that spans several recent stores waits until they reach the cache. +*/ +NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept +{ +#if NLOHMANN_VIEW_NEON + namespace dtoa = ::nlohmann::detail::dtoa_impl; + const std::uint64_t upper = digits / 100000000u; + const std::uint64_t b0 = upper / 100000000u; // one digit + const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u); + const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u); + // leading and trailing zero digits (as dtoa_impl::write_decimal()) + int leading = 7; + if (b0 == 0) + { + leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8); + } + int zeros = 16; + if (b2 != 0) + { + zeros = count_trailing_zeros(b2) / 8; + } + else if (b1 != 0) + { + zeros = 8 + (count_trailing_zeros(b1) / 8); + } + const int k = 24 - leading - zeros; // significant digits + const int n = k + exp + zeros; // position of the point after the first digit + if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15)) + { + static const std::array iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}}; + const int pad = n <= 0 ? 1 - n : 0; + const int len = k + pad; + const int point = n + pad; + // the 24 digit bytes in memory order, then '0's + const std::uint64_t zero_chars = 0x3030303030303030u; + const uint8x16x2_t table = {{ + vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))), + vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0')) + } + }; + const uint8x16_t s = vdupq_n_u8(static_cast(leading - pad)); + const uint8x16_t at_point = vdupq_n_u8(static_cast(point)); + for (std::size_t half = 0; half < 2; ++half) + { + const uint8x16_t i = vld1q_u8(iota.data() + (16 * half)); + // (+ 0xFF is - 1 after the point; indexes past the digits read a '0') + const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31)); + vst1q_u8(reinterpret_cast(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0" + } +#endif + return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp); +} + /*! @brief write a view's subtree as basic_json::dump() writes the value @@ -5796,10 +5861,15 @@ class view_serializer room(n->len); copy(src + n->off, n->len); } + else if (std::is_same::value) + { + room(64); + w = write_double_at(w, *n); + } else { m_out.set_cursor(w); - write_float(float_value(m_doc, *n)); + write_float_node(*n); w = m_out.cursor(); lim = m_out.limit(); } @@ -5966,7 +6036,7 @@ class view_serializer } else { - write_float(float_value(m_doc, n)); + write_float_node(n); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) @@ -5978,6 +6048,83 @@ class view_serializer } } + /// a float node as dump() writes it + void write_float_node(const node& n) + { + write_float_node(n, std::is_same {}); + } + + void write_float_node(const node& n, std::false_type /*other*/) + { + write_float(float_value(m_doc, n)); + } + + void write_float_node(const node& n, std::true_type /*double*/) + { + m_out.reserve(64); + m_out.set_cursor(write_double_at(m_out.cursor(), n)); + } + + /*! + @brief (doubles) the float at n as dump() writes it, at w (64 bytes of room) + + A token of at most 15 significant digits is written from its digits, + without a conversion: two decimals of at most 15 digits are farther + apart than the rounding interval of a (normal) double (the argument + behind DBL_DIG), so the token's digits are the shortest ones of its + double, which the library's conversion writes (Zmij). Other tokens are + converted from the digits already read. + */ + char* write_double_at(char* w, const node& n) + { + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if ((n.flags & node_flags::storage) != node_flags::edited && int_digits + frac_digits <= 19) + { + const auto* const first = reinterpret_cast(m_doc.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const float_significand d = layout_decimal(first, first + n.len, int_digits, frac_digits, reinterpret_cast(m_doc.src + m_doc.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + // (the exponent keeps the value far from subnormals and overflow) + if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290) + { + *w = '-'; + return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast(d.exponent)); + } + return write_double_value_at(w, decimal_to_float(d)); // (without reading the token again) + } + return write_double_value_at(w, static_cast(float_value(m_doc, n))); + } + + /// n bytes of text at w + static char* write_text_at(char* w, const char* text, std::size_t n) noexcept + { + std::memcpy(w, text, n); + return w + n; + } + + /// a double as dump() writes it, at w (64 bytes of room) + static char* write_double_value_at(char* w, double x) + { + if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x))) + { + return write_text_at(w, "null", 4); + } +#if NLOHMANN_VIEW_NEON + std::uint64_t bits = 0; + std::memcpy(&bits, &x, sizeof(bits)); + *w = '-'; + w += bits >> 63u; + bits &= ~(std::uint64_t{1} << 63u); + if (bits == 0) + { + return write_text_at(w, "0.0", 3); + } + const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits); + return write_decimal(w, d.significand, d.exponent); +#else + return ::nlohmann::detail::to_chars(w, w + 64, x); +#endif + } + /// as serializer::dump_float() void write_float(number_float_t x) { diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index b8d124d59..8eb7bdcd0 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1151,6 +1151,47 @@ TEST_CASE("json_view dump") CHECK(d.root().dump(0, ' ', false, json_view::number_format::source) == "[\n1.50,\n1E2,\n-0,\n-0.0,\n123456789012345678901234567890,\n18446744073709551615,\n-9223372036854775808,\n0.1,\n1e-7,\n5e-324\n]"); CHECK(d.root().dump(-1, ' ', true, json_view::number_format::source) == "[1.50,1E2,-0,-0.0,123456789012345678901234567890,18446744073709551615,-9223372036854775808,0.1,1e-7,5e-324]"); + // float tokens of up to 17 significant digits in every spelling: those + // of at most 15 digits are written from their digits, the others + // through the conversion; both as dump() writes them + { + std::mt19937_64 tokens(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + std::string many_tokens = "["; + for (int i = 0; i < 20000; ++i) + { + const auto length = static_cast(1 + (tokens() % 17)); + std::string digits(1, static_cast('1' + (tokens() % 9))); + for (std::size_t k = 1; k < length; ++k) + { + digits += static_cast('0' + (tokens() % 10)); + } + digits += std::string(tokens() % 4, '0'); // trailing zeros + std::string token = tokens() % 3 == 0 ? "-" : ""; + const auto point = static_cast(tokens() % (digits.size() + 1)); + if (point == 0) + { + token += "0." + std::string(tokens() % 5, '0') + digits; + } + else + { + token += digits.substr(0, point) + (point < digits.size() ? "." + digits.substr(point) : ""); + } + // an exponent that keeps the value between about 1e-320 and 1e300 + const int exponent = static_cast(tokens() % 600) - 300 - static_cast(point); + if (tokens() % 4 != 0) + { + token += (tokens() % 2 == 0 ? "e" : "E") + std::string(exponent >= 0 && tokens() % 2 == 0 ? "+" : "") + std::to_string(exponent); + } + else if (point == digits.size()) + { + token += ".0"; // (a float, not an integer) + } + many_tokens += (i != 0 ? "," : "") + token; + } + many_tokens += ']'; + CHECK(json_document::parse(many_tokens).root().dump() == json::parse(many_tokens).dump()); + } + // random doubles, written as parse() and dump() would std::mt19937_64 rng(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) std::string many = "["; diff --git a/tests/src/unit-json_view_edit.cpp b/tests/src/unit-json_view_edit.cpp index da65cde3b..a01b7c13b 100644 --- a/tests/src/unit-json_view_edit.cpp +++ b/tests/src/unit-json_view_edit.cpp @@ -26,6 +26,7 @@ using ptr_t = ordered_json::json_pointer; #include #include #include +#include #include #include #include @@ -481,6 +482,19 @@ TEST_CASE("json_view edits: views and values") CHECK(d.root().materialize().dump() == json::parse(R"([1.5, 100.0, 0.1, null, null, 18446744073709551615, -9223372036854775808])").dump()); } + SECTION("numbers of other float types") + { + // doubles have their own path to the output; other float types are + // written as basic_json writes them, non-finite values as null + using json_float = nlohmann::basic_json; + using document_float = nlohmann::basic_json_document; + document_float d = document_float::parse("[1.5]"); + d.push_back(d.root(), std::numeric_limits::quiet_NaN()); + d.push_back(d.root(), -std::numeric_limits::infinity()); + CHECK(d.root().dump() == "[1.5,null,null]"); + CHECK(d.root().dump(2) == json_float::parse("[1.5, null, null]").dump(2)); + } + SECTION("nulls become containers, and the root can be replaced") { json_editable_document d = json_editable_document::parse("[null, null]");