From bf6b5b47192f3cc3cc398161fcb6a00beaeaa5ea Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Mon, 28 Sep 2026 23:24:57 +0200 Subject: [PATCH] Convert the floats of json_view from the digit layout The parser records where the integer digits, the fraction digits, and the exponent of a float token are. For doubles with at most 19 digits, the value is now read from that layout: the digits eight at a time, without scanning the token, and rounded with Clinger's fast path where both operands are exact, else with the Eisel-Lemire algorithm (which needs no fallback for up to 19 digits). Both round correctly, so the values are those of parse(); other tokens and types keep the library's conversion. get(), materialize(), dump(), and comparisons use it. Traversing canada.json (111,000 floats, every number converted): 0.53 -> 0.86 GB/s. Tests add tokens around the limits (19 and 20 digits, 2^53, 10^22) to the bit-for-bit comparison with parse(). Signed-off-by: Niels Lohmann --- include/nlohmann/detail/view/materialize.hpp | 2 +- include/nlohmann/detail/view/number.hpp | 97 +++++++++++++++++ include/nlohmann/detail/view/serializer.hpp | 2 +- include/nlohmann/detail/view/value.hpp | 2 +- single_include/nlohmann/json_view.hpp | 105 ++++++++++++++++++- tests/src/unit-json_view.cpp | 7 +- 6 files changed, 208 insertions(+), 7 deletions(-) diff --git a/include/nlohmann/detail/view/materialize.hpp b/include/nlohmann/detail/view/materialize.hpp index e9ffe0668..79fafee1b 100644 --- a/include/nlohmann/detail/view/materialize.hpp +++ b/include/nlohmann/detail/view/materialize.hpp @@ -81,7 +81,7 @@ BasicJsonType materialize(const document_data& d, const node* n) ++n; break; case value_t::number_float: - sax.number_float(float_value(d.str(*n), *n), no_token); + sax.number_float(float_value(d, *n), no_token); ++n; break; case value_t::boolean: diff --git a/include/nlohmann/detail/view/number.hpp b/include/nlohmann/detail/view/number.hpp index 9ccce1988..359f4158e 100644 --- a/include/nlohmann/detail/view/number.hpp +++ b/include/nlohmann/detail/view/number.hpp @@ -8,12 +8,19 @@ #pragma once +#include // array +#include // FLT_EVAL_METHOD #include // size_t +#include // int64_t, uint64_t +#include // memcpy #include // string +#include // integral_constant, is_same #include +#include #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -68,6 +75,96 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) return v; } +/*! +@brief the double of a float token with at most 19 digits, from its layout + +The digit layout recorded while parsing says where the integer digits, the +fraction digits, and the exponent are, so the digits are read eight at a +time without scanning. The result is correctly rounded (Clinger's fast path +where both operands are exact, else the Eisel-Lemire algorithm, which needs +no fallback for up to 19 digits), so it is the value parse() produces. + +@param[in] p first character of the token +@param[in] e end of the token +@param[in] limit end of the readable memory (the source text) +*/ +NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + const bool negative = *p == '-'; + p += negative ? 1 : 0; + std::uint64_t w = parse_upto19(p, int_digits, limit); + p += int_digits; + std::int64_t q = 0; + if (frac_digits != 0) + { + w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit); + p += 1 + frac_digits; + q = -static_cast(frac_digits); + } + if (p != e) + { + // [eE][+-]digits; huge exponents saturate (the parser rejected overflow) + ++p; + const bool exp_negative = *p == '-'; + p += (*p == '-' || *p == '+') ? 1 : 0; + std::int64_t exp_value = 0; + for (; p != e; ++p) + { + if (exp_value < 0x10000000) + { + exp_value = (exp_value * 10) + (*p - '0'); + } + } + q += exp_negative ? -exp_value : exp_value; + } + + double result = 0; + if (w != 0) + { +#if !defined(FLT_EVAL_METHOD) || FLT_EVAL_METHOD == 0 + static const std::array pow10 = {{1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22}}; + if (q >= -22 && q <= 22 && w <= (std::uint64_t{1} << 53)) + { + // Clinger's fast path: both operands exact, one rounding + result = static_cast(w); + result = q < 0 ? result / pow10[static_cast(-q)] : result * pow10[static_cast(q)]; + return negative ? -result : result; + } +#endif + const std::uint64_t bits = eisel_lemire(q, w); + std::memcpy(&result, &bits, sizeof(result)); + } + return negative ? -result : result; +} + +/// the value of the float token of a node, as parse() converts it; doubles +/// with at most 19 digits are converted from the digit layout +template +FloatType float_value(const document_data& d, const node& n) +{ + return float_value(d, n, std::is_same {}); +} + +template +FloatType float_value(const document_data& d, const node& n, std::true_type /*double*/) +{ + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many") + { + // (a float token not written by an edit is in the text) + const auto* const first = reinterpret_cast(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return layout_double(first, first + n.len, int_digits, frac_digits, reinterpret_cast(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return float_value(d.str(n), n); +} + +template +FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/) +{ + return float_value(d.str(n), n); +} + } // namespace view } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index 98c7335dd..26bfb3225 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -256,7 +256,7 @@ class view_serializer } else { - write_float(float_value(m_doc.str(n), n)); + write_float(float_value(m_doc, n)); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) diff --git a/include/nlohmann/detail/view/value.hpp b/include/nlohmann/detail/view/value.hpp index b43a9c4df..23d291634 100644 --- a/include/nlohmann/detail/view/value.hpp +++ b/include/nlohmann/detail/view/value.hpp @@ -49,7 +49,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE T arithmetic_value(const document_data& d, const nod case value_t::number_integer: return static_cast(static_cast(static_cast(integer_bits(n)))); case value_t::number_float: - return static_cast(float_value(d.str(n), n)); + return static_cast(float_value(d, n)); case value_t::boolean: return static_cast((n.flags & node_flags::is_true) != 0); case value_t::null: diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index c6ddda728..393b52c75 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -2523,14 +2523,23 @@ NLOHMANN_JSON_NAMESPACE_END +#include // array +#include // FLT_EVAL_METHOD #include // size_t +#include // int64_t, uint64_t +#include // memcpy #include // string +#include // integral_constant, is_same // #include +// #include + // #include // #include +// #include + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -2585,6 +2594,96 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) return v; } +/*! +@brief the double of a float token with at most 19 digits, from its layout + +The digit layout recorded while parsing says where the integer digits, the +fraction digits, and the exponent are, so the digits are read eight at a +time without scanning. The result is correctly rounded (Clinger's fast path +where both operands are exact, else the Eisel-Lemire algorithm, which needs +no fallback for up to 19 digits), so it is the value parse() produces. + +@param[in] p first character of the token +@param[in] e end of the token +@param[in] limit end of the readable memory (the source text) +*/ +NLOHMANN_VIEW_ALWAYS_INLINE double layout_double(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + const bool negative = *p == '-'; + p += negative ? 1 : 0; + std::uint64_t w = parse_upto19(p, int_digits, limit); + p += int_digits; + std::int64_t q = 0; + if (frac_digits != 0) + { + w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit); + p += 1 + frac_digits; + q = -static_cast(frac_digits); + } + if (p != e) + { + // [eE][+-]digits; huge exponents saturate (the parser rejected overflow) + ++p; + const bool exp_negative = *p == '-'; + p += (*p == '-' || *p == '+') ? 1 : 0; + std::int64_t exp_value = 0; + for (; p != e; ++p) + { + if (exp_value < 0x10000000) + { + exp_value = (exp_value * 10) + (*p - '0'); + } + } + q += exp_negative ? -exp_value : exp_value; + } + + double result = 0; + if (w != 0) + { +#if !defined(FLT_EVAL_METHOD) || FLT_EVAL_METHOD == 0 + static const std::array pow10 = {{1e0, 1e1, 1e2, 1e3, 1e4, 1e5, 1e6, 1e7, 1e8, 1e9, 1e10, 1e11, 1e12, 1e13, 1e14, 1e15, 1e16, 1e17, 1e18, 1e19, 1e20, 1e21, 1e22}}; + if (q >= -22 && q <= 22 && w <= (std::uint64_t{1} << 53)) + { + // Clinger's fast path: both operands exact, one rounding + result = static_cast(w); + result = q < 0 ? result / pow10[static_cast(-q)] : result * pow10[static_cast(q)]; + return negative ? -result : result; + } +#endif + const std::uint64_t bits = eisel_lemire(q, w); + std::memcpy(&result, &bits, sizeof(result)); + } + return negative ? -result : result; +} + +/// the value of the float token of a node, as parse() converts it; doubles +/// with at most 19 digits are converted from the digit layout +template +FloatType float_value(const document_data& d, const node& n) +{ + return float_value(d, n, std::is_same {}); +} + +template +FloatType float_value(const document_data& d, const node& n, std::true_type /*double*/) +{ + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many") + { + // (a float token not written by an edit is in the text) + const auto* const first = reinterpret_cast(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return layout_double(first, first + n.len, int_digits, frac_digits, reinterpret_cast(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return float_value(d.str(n), n); +} + +template +FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/) +{ + return float_value(d.str(n), n); +} + } // namespace view } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -2653,7 +2752,7 @@ BasicJsonType materialize(const document_data& d, const node* n) ++n; break; case value_t::number_float: - sax.number_float(float_value(d.str(*n), *n), no_token); + sax.number_float(float_value(d, *n), no_token); ++n; break; case value_t::boolean: @@ -3142,7 +3241,7 @@ class view_serializer } else { - write_float(float_value(m_doc.str(n), n)); + write_float(float_value(m_doc, n)); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) @@ -3475,7 +3574,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE T arithmetic_value(const document_data& d, const nod case value_t::number_integer: return static_cast(static_cast(static_cast(integer_bits(n)))); case value_t::number_float: - return static_cast(float_value(d.str(n), n)); + return static_cast(float_value(d, n)); case value_t::boolean: return static_cast((n.flags & node_flags::is_true) != 0); case value_t::null: diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index 4e7e409d5..a6d34d5a1 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -826,7 +826,12 @@ TEST_CASE("json_view values") std::mt19937_64 rng(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) std::vector tokens = {"0.1", "-0.0", "1e308", "1.7976931348623157e308", "2.2250738585072011e-308", "4.9e-324", "5e-324", "0.1000000000000000055511151231257827021181583404541015625", "123456789012345678901234567890", - "9007199254740993", "1.00000000000000011102230246251565404236316680908203125", "7.2057594037927933e16" + "9007199254740993", "1.00000000000000011102230246251565404236316680908203125", "7.2057594037927933e16", + // around the limits of the conversion from the digit layout: 19 and 20 + // digits, and those of Clinger's fast path (2^53, 10^22) + "1234567890.123456789", "1234567890.1234567891", "0.0000000000000000001", "123456789012345678.9", + "9007199254740992.0", "9007199254740993.0", "9007199254740994.0", "1.5e22", "1.5e23", "15e-22", "15e-23", + "1e-400", "0.0e0", "-0.0e-5", "12E+3", "12e-0" }; for (int i = 0; i < 20000; ++i) {