diff --git a/include/nlohmann/detail/view/number.hpp b/include/nlohmann/detail/view/number.hpp index f64573582..af02bd47c 100644 --- a/include/nlohmann/detail/view/number.hpp +++ b/include/nlohmann/detail/view/number.hpp @@ -109,20 +109,26 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) return v; } +/// the significand (at most 19 digits) and the decimal exponent of a float token +struct token_decimal +{ + std::uint64_t w; + std::int64_t q; + bool negative; +}; + /*! -@brief the double of a float token with at most 19 digits, from its layout +@brief the digits of a float token with at most 19 digits, from its layout The digit layout recorded while parsing says where the integer digits, the fraction digits, and the exponent are, so the digits are read eight at a -time without scanning. The result is correctly rounded (Clinger's fast path -where both operands are exact, else the Eisel-Lemire algorithm, which needs -no fallback for up to 19 digits), so it is the value parse() produces. +time without scanning. @param[in] p first character of the token @param[in] e end of the token @param[in] limit end of the readable memory (the source text) */ -NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +NLOHMANN_VIEW_ALWAYS_INLINE token_decimal layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept { const bool negative = *p == '-'; p += negative ? 1 : 0; @@ -156,6 +162,21 @@ NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const u q += exp_negative ? -exp_value : exp_value; } + return token_decimal{w, q, negative}; +} + +/*! +@brief the double of the digits of a float token (at most 19 digits) + +The result is correctly rounded (Clinger's fast path where both operands are +exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19 +digits), so it is the value parse() produces. +*/ +NLOHMANN_VIEW_ALWAYS_INLINE double decimal_to_double(const token_decimal& d) noexcept +{ + const std::uint64_t w = d.w; + const std::int64_t q = d.q; + const bool negative = d.negative; double result = 0; if (w != 0) { @@ -175,6 +196,12 @@ NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const u return negative ? -result : result; } +/// the double of a float token with at most 19 digits, from its layout +NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + return decimal_to_double(layout_decimal(p, e, int_digits, frac_digits, limit)); +} + /// the value of a float set by an edit: its token (the shortest round-trip /// text, or "nan", "inf", "-inf") in the edit arena template diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index 42c904b83..4716a9be8 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -23,6 +23,7 @@ #include #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -221,6 +222,69 @@ struct dump_style bool source_numbers = false; ///< copy number tokens from the source }; +/*! +@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0, +at most 17 digits; up to 41 bytes are written at first) + +With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in +vector registers: output byte i is byte s + i of the digits (after '0's) +before the point, and byte s + i - 1 after it. The portable code writes the +digits to a buffer and copies them from there at another offset, and a load +that spans several recent stores waits until they reach the cache. +*/ +NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept +{ +#if NLOHMANN_VIEW_NEON + namespace dtoa = ::nlohmann::detail::dtoa_impl; + const std::uint64_t upper = digits / 100000000u; + const std::uint64_t b0 = upper / 100000000u; // one digit + const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u); + const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u); + // leading and trailing zero digits (as dtoa_impl::write_decimal()) + int leading = 7; + if (b0 == 0) + { + leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8); + } + int zeros = 16; + if (b2 != 0) + { + zeros = count_trailing_zeros(b2) / 8; + } + else if (b1 != 0) + { + zeros = 8 + (count_trailing_zeros(b1) / 8); + } + const int k = 24 - leading - zeros; // significant digits + const int n = k + exp + zeros; // position of the point after the first digit + if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15)) + { + static const std::array iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}}; + const int pad = n <= 0 ? 1 - n : 0; + const int len = k + pad; + const int point = n + pad; + // the 24 digit bytes in memory order, then '0's + const std::uint64_t zero_chars = 0x3030303030303030u; + const uint8x16x2_t table = {{ + vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))), + vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0')) + } + }; + const uint8x16_t s = vdupq_n_u8(static_cast(leading - pad)); + const uint8x16_t at_point = vdupq_n_u8(static_cast(point)); + for (std::size_t half = 0; half < 2; ++half) + { + const uint8x16_t i = vld1q_u8(iota.data() + (16 * half)); + // (+ 0xFF is - 1 after the point; indexes past the digits read a '0') + const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31)); + vst1q_u8(reinterpret_cast(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0" + } +#endif + return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp); +} + /*! @brief write a view's subtree as basic_json::dump() writes the value @@ -496,10 +560,15 @@ class view_serializer room(n->len); copy(src + n->off, n->len); } + else if (std::is_same::value) + { + room(64); + w = write_double_at(w, *n); + } else { m_out.set_cursor(w); - write_float(float_value(m_doc, *n)); + write_float_node(*n); w = m_out.cursor(); lim = m_out.limit(); } @@ -666,7 +735,7 @@ class view_serializer } else { - write_float(float_value(m_doc, n)); + write_float_node(n); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) @@ -678,6 +747,83 @@ class view_serializer } } + /// a float node as dump() writes it + void write_float_node(const node& n) + { + write_float_node(n, std::is_same {}); + } + + void write_float_node(const node& n, std::false_type /*other*/) + { + write_float(float_value(m_doc, n)); + } + + void write_float_node(const node& n, std::true_type /*double*/) + { + m_out.reserve(64); + m_out.set_cursor(write_double_at(m_out.cursor(), n)); + } + + /*! + @brief (doubles) the float at n as dump() writes it, at w (64 bytes of room) + + A token of at most 15 significant digits is written from its digits, + without a conversion: two decimals of at most 15 digits are farther + apart than the rounding interval of a (normal) double (the argument + behind DBL_DIG), so the token's digits are the shortest ones of its + double, which the library's conversion writes (Zmij). Other tokens are + converted from the digits already read. + */ + char* write_double_at(char* w, const node& n) + { + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if ((n.flags & node_flags::storage) != node_flags::edited && int_digits + frac_digits <= 19) + { + const auto* const first = reinterpret_cast(m_doc.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const token_decimal d = layout_decimal(first, first + n.len, int_digits, frac_digits, reinterpret_cast(m_doc.src + m_doc.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + // (the exponent keeps the value far from subnormals and overflow) + if (d.w != 0 && d.w < 1000000000000000u && d.q >= -290 && d.q <= 290) + { + *w = '-'; + return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast(d.q)); + } + return write_double_value_at(w, decimal_to_double(d)); // (without reading the token again) + } + return write_double_value_at(w, static_cast(float_value(m_doc, n))); + } + + /// n bytes of text at w + static char* write_text_at(char* w, const char* text, std::size_t n) noexcept + { + std::memcpy(w, text, n); + return w + n; + } + + /// a double as dump() writes it, at w (64 bytes of room) + static char* write_double_value_at(char* w, double x) + { + if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x))) + { + return write_text_at(w, "null", 4); + } +#if NLOHMANN_VIEW_NEON + std::uint64_t bits = 0; + std::memcpy(&bits, &x, sizeof(bits)); + *w = '-'; + w += bits >> 63u; + bits &= ~(std::uint64_t{1} << 63u); + if (bits == 0) + { + return write_text_at(w, "0.0", 3); + } + const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits); + return write_decimal(w, d.significand, d.exponent); +#else + return ::nlohmann::detail::to_chars(w, w + 64, x); +#endif + } + /// as serializer::dump_float() void write_float(number_float_t x) { diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 88ac74e1f..0e9412854 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -3902,20 +3902,26 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) return v; } +/// the significand (at most 19 digits) and the decimal exponent of a float token +struct token_decimal +{ + std::uint64_t w; + std::int64_t q; + bool negative; +}; + /*! -@brief the double of a float token with at most 19 digits, from its layout +@brief the digits of a float token with at most 19 digits, from its layout The digit layout recorded while parsing says where the integer digits, the fraction digits, and the exponent are, so the digits are read eight at a -time without scanning. The result is correctly rounded (Clinger's fast path -where both operands are exact, else the Eisel-Lemire algorithm, which needs -no fallback for up to 19 digits), so it is the value parse() produces. +time without scanning. @param[in] p first character of the token @param[in] e end of the token @param[in] limit end of the readable memory (the source text) */ -NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +NLOHMANN_VIEW_ALWAYS_INLINE token_decimal layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept { const bool negative = *p == '-'; p += negative ? 1 : 0; @@ -3949,6 +3955,21 @@ NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const u q += exp_negative ? -exp_value : exp_value; } + return token_decimal{w, q, negative}; +} + +/*! +@brief the double of the digits of a float token (at most 19 digits) + +The result is correctly rounded (Clinger's fast path where both operands are +exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19 +digits), so it is the value parse() produces. +*/ +NLOHMANN_VIEW_ALWAYS_INLINE double decimal_to_double(const token_decimal& d) noexcept +{ + const std::uint64_t w = d.w; + const std::int64_t q = d.q; + const bool negative = d.negative; double result = 0; if (w != 0) { @@ -3968,6 +3989,12 @@ NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const u return negative ? -result : result; } +/// the double of a float token with at most 19 digits, from its layout +NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + return decimal_to_double(layout_decimal(p, e, int_digits, frac_digits, limit)); +} + /// the value of a float set by an edit: its token (the shortest round-trip /// text, or "nan", "inf", "-inf") in the edit arena template @@ -5317,6 +5344,8 @@ NLOHMANN_JSON_NAMESPACE_END // #include +// #include + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -5515,6 +5544,69 @@ struct dump_style bool source_numbers = false; ///< copy number tokens from the source }; +/*! +@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0, +at most 17 digits; up to 41 bytes are written at first) + +With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in +vector registers: output byte i is byte s + i of the digits (after '0's) +before the point, and byte s + i - 1 after it. The portable code writes the +digits to a buffer and copies them from there at another offset, and a load +that spans several recent stores waits until they reach the cache. +*/ +NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept +{ +#if NLOHMANN_VIEW_NEON + namespace dtoa = ::nlohmann::detail::dtoa_impl; + const std::uint64_t upper = digits / 100000000u; + const std::uint64_t b0 = upper / 100000000u; // one digit + const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u); + const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u); + // leading and trailing zero digits (as dtoa_impl::write_decimal()) + int leading = 7; + if (b0 == 0) + { + leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8); + } + int zeros = 16; + if (b2 != 0) + { + zeros = count_trailing_zeros(b2) / 8; + } + else if (b1 != 0) + { + zeros = 8 + (count_trailing_zeros(b1) / 8); + } + const int k = 24 - leading - zeros; // significant digits + const int n = k + exp + zeros; // position of the point after the first digit + if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15)) + { + static const std::array iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}}; + const int pad = n <= 0 ? 1 - n : 0; + const int len = k + pad; + const int point = n + pad; + // the 24 digit bytes in memory order, then '0's + const std::uint64_t zero_chars = 0x3030303030303030u; + const uint8x16x2_t table = {{ + vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))), + vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0')) + } + }; + const uint8x16_t s = vdupq_n_u8(static_cast(leading - pad)); + const uint8x16_t at_point = vdupq_n_u8(static_cast(point)); + for (std::size_t half = 0; half < 2; ++half) + { + const uint8x16_t i = vld1q_u8(iota.data() + (16 * half)); + // (+ 0xFF is - 1 after the point; indexes past the digits read a '0') + const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31)); + vst1q_u8(reinterpret_cast(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0" + } +#endif + return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp); +} + /*! @brief write a view's subtree as basic_json::dump() writes the value @@ -5790,10 +5882,15 @@ class view_serializer room(n->len); copy(src + n->off, n->len); } + else if (std::is_same::value) + { + room(64); + w = write_double_at(w, *n); + } else { m_out.set_cursor(w); - write_float(float_value(m_doc, *n)); + write_float_node(*n); w = m_out.cursor(); lim = m_out.limit(); } @@ -5960,7 +6057,7 @@ class view_serializer } else { - write_float(float_value(m_doc, n)); + write_float_node(n); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) @@ -5972,6 +6069,83 @@ class view_serializer } } + /// a float node as dump() writes it + void write_float_node(const node& n) + { + write_float_node(n, std::is_same {}); + } + + void write_float_node(const node& n, std::false_type /*other*/) + { + write_float(float_value(m_doc, n)); + } + + void write_float_node(const node& n, std::true_type /*double*/) + { + m_out.reserve(64); + m_out.set_cursor(write_double_at(m_out.cursor(), n)); + } + + /*! + @brief (doubles) the float at n as dump() writes it, at w (64 bytes of room) + + A token of at most 15 significant digits is written from its digits, + without a conversion: two decimals of at most 15 digits are farther + apart than the rounding interval of a (normal) double (the argument + behind DBL_DIG), so the token's digits are the shortest ones of its + double, which the library's conversion writes (Zmij). Other tokens are + converted from the digits already read. + */ + char* write_double_at(char* w, const node& n) + { + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if ((n.flags & node_flags::storage) != node_flags::edited && int_digits + frac_digits <= 19) + { + const auto* const first = reinterpret_cast(m_doc.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const token_decimal d = layout_decimal(first, first + n.len, int_digits, frac_digits, reinterpret_cast(m_doc.src + m_doc.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + // (the exponent keeps the value far from subnormals and overflow) + if (d.w != 0 && d.w < 1000000000000000u && d.q >= -290 && d.q <= 290) + { + *w = '-'; + return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast(d.q)); + } + return write_double_value_at(w, decimal_to_double(d)); // (without reading the token again) + } + return write_double_value_at(w, static_cast(float_value(m_doc, n))); + } + + /// n bytes of text at w + static char* write_text_at(char* w, const char* text, std::size_t n) noexcept + { + std::memcpy(w, text, n); + return w + n; + } + + /// a double as dump() writes it, at w (64 bytes of room) + static char* write_double_value_at(char* w, double x) + { + if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x))) + { + return write_text_at(w, "null", 4); + } +#if NLOHMANN_VIEW_NEON + std::uint64_t bits = 0; + std::memcpy(&bits, &x, sizeof(bits)); + *w = '-'; + w += bits >> 63u; + bits &= ~(std::uint64_t{1} << 63u); + if (bits == 0) + { + return write_text_at(w, "0.0", 3); + } + const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits); + return write_decimal(w, d.significand, d.exponent); +#else + return ::nlohmann::detail::to_chars(w, w + 64, x); +#endif + } + /// as serializer::dump_float() void write_float(number_float_t x) { diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 53476d62b..073126e8d 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -1093,6 +1093,47 @@ TEST_CASE("json_view dump") CHECK(d.root().dump(0, ' ', false, json_view::number_format::source) == "[\n1.50,\n1E2,\n-0,\n-0.0,\n123456789012345678901234567890,\n18446744073709551615,\n-9223372036854775808,\n0.1,\n1e-7,\n5e-324\n]"); CHECK(d.root().dump(-1, ' ', true, json_view::number_format::source) == "[1.50,1E2,-0,-0.0,123456789012345678901234567890,18446744073709551615,-9223372036854775808,0.1,1e-7,5e-324]"); + // float tokens of up to 17 significant digits in every spelling: those + // of at most 15 digits are written from their digits, the others + // through the conversion; both as dump() writes them + { + std::mt19937_64 tokens(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + std::string many_tokens = "["; + for (int i = 0; i < 20000; ++i) + { + const auto length = static_cast(1 + (tokens() % 17)); + std::string digits(1, static_cast('1' + (tokens() % 9))); + for (std::size_t k = 1; k < length; ++k) + { + digits += static_cast('0' + (tokens() % 10)); + } + digits += std::string(tokens() % 4, '0'); // trailing zeros + std::string token = tokens() % 3 == 0 ? "-" : ""; + const auto point = static_cast(tokens() % (digits.size() + 1)); + if (point == 0) + { + token += "0." + std::string(tokens() % 5, '0') + digits; + } + else + { + token += digits.substr(0, point) + (point < digits.size() ? "." + digits.substr(point) : ""); + } + // an exponent that keeps the value between about 1e-320 and 1e300 + const int exponent = static_cast(tokens() % 600) - 300 - static_cast(point); + if (tokens() % 4 != 0) + { + token += (tokens() % 2 == 0 ? "e" : "E") + std::string(exponent >= 0 && tokens() % 2 == 0 ? "+" : "") + std::to_string(exponent); + } + else if (point == digits.size()) + { + token += ".0"; // (a float, not an integer) + } + many_tokens += (i != 0 ? "," : "") + token; + } + many_tokens += ']'; + CHECK(json_document::parse(many_tokens).root().dump() == json::parse(many_tokens).dump()); + } + // random doubles, written as parse() and dump() would std::mt19937_64 rng(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) std::string many = "["; diff --git a/tests/src/unit-json_view_edit.cpp b/tests/src/unit-json_view_edit.cpp index c865dddd7..4bf016b89 100644 --- a/tests/src/unit-json_view_edit.cpp +++ b/tests/src/unit-json_view_edit.cpp @@ -26,6 +26,7 @@ using ptr_t = ordered_json::json_pointer; #include #include #include +#include #include #include #include @@ -474,6 +475,19 @@ TEST_CASE("json_view edits: views and values") CHECK(d.root().materialize().dump() == json::parse(R"([1.5, 100.0, 0.1, null, null, 18446744073709551615, -9223372036854775808])").dump()); } + SECTION("numbers of other float types") + { + // doubles have their own path to the output; other float types are + // written as basic_json writes them, non-finite values as null + using json_float = nlohmann::basic_json; + using document_float = nlohmann::basic_json_document; + document_float d = document_float::parse("[1.5]"); + d.push_back(d.root(), std::numeric_limits::quiet_NaN()); + d.push_back(d.root(), -std::numeric_limits::infinity()); + CHECK(d.root().dump() == "[1.5,null,null]"); + CHECK(d.root().dump(2) == json_float::parse("[1.5, null, null]").dump(2)); + } + SECTION("nulls become containers, and the root can be replaced") { json_editable_document d = json_editable_document::parse("[null, null]");