diff --git a/include/nlohmann/detail/view/materialize.hpp b/include/nlohmann/detail/view/materialize.hpp index e9ffe0668..79fafee1b 100644 --- a/include/nlohmann/detail/view/materialize.hpp +++ b/include/nlohmann/detail/view/materialize.hpp @@ -81,7 +81,7 @@ BasicJsonType materialize(const document_data& d, const node* n) ++n; break; case value_t::number_float: - sax.number_float(float_value(d.str(*n), *n), no_token); + sax.number_float(float_value(d, *n), no_token); ++n; break; case value_t::boolean: diff --git a/include/nlohmann/detail/view/number.hpp b/include/nlohmann/detail/view/number.hpp index c9befdd49..24a87a7b8 100644 --- a/include/nlohmann/detail/view/number.hpp +++ b/include/nlohmann/detail/view/number.hpp @@ -9,11 +9,15 @@ #pragma once #include // size_t +#include // int64_t, uint64_t #include // string +#include // integral_constant #include +#include #include #include +#include NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -62,6 +66,96 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) return convert_float(first, last, dot, mantissa_end); } +/*! +@brief the digits of a float token with at most 19 digits, from its layout + +The digit layout recorded while parsing says where the integer digits, the +fraction digits, and the exponent are, so the digits are read eight at a +time without scanning. + +@param[in] p first character of the token +@param[in] e end of the token +@param[in] limit end of the readable memory (the source text) +*/ +NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + const bool negative = *p == '-'; + p += negative ? 1 : 0; + std::uint64_t w = parse_upto19(p, int_digits, limit); + p += int_digits; + std::int64_t q = 0; + if (frac_digits != 0) + { + w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit); + p += 1 + frac_digits; + q = -static_cast(frac_digits); + } + if (p != e) + { + // [eE][+-]digits; huge exponents saturate (the parser rejected overflow) + ++p; + const bool exp_negative = *p == '-'; + p += (*p == '-' || *p == '+') ? 1 : 0; + std::int64_t exp_value = 0; + for (; p != e; ++p) + { + if (exp_value < 0x10000000) + { + exp_value = (exp_value * 10) + (*p - '0'); + } + } + q += exp_negative ? -exp_value : exp_value; + } + + float_significand d; + d.w = w; + d.exponent = q; + d.negative = negative; + return d; +} + +/*! +@brief the value of a float token with at most 19 digits, from its layout + +The result is correctly rounded by the lexer's conversion +(detail::decimal_to_float(): Clinger's fast path where both operands are +exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19 +digits), so it is the value parse() produces. +*/ +template +NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + return decimal_to_float(layout_decimal(p, e, int_digits, frac_digits, limit)); +} + +/// the value of the float token of a node, as parse() converts it; floats and +/// doubles with at most 19 digits are converted from the digit layout +template +FloatType float_value(const document_data& d, const node& n) +{ + return float_value(d, n, std::integral_constant::value> {}); +} + +template +FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/) +{ + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many") + { + // (a float token not written by an edit is in the text) + const auto* const first = reinterpret_cast(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return layout_float(first, first + n.len, int_digits, frac_digits, reinterpret_cast(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return float_value(d.str(n), n); +} + +template +FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/) +{ + return float_value(d.str(n), n); +} + } // namespace view } // namespace detail NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/serializer.hpp b/include/nlohmann/detail/view/serializer.hpp index 98c7335dd..26bfb3225 100644 --- a/include/nlohmann/detail/view/serializer.hpp +++ b/include/nlohmann/detail/view/serializer.hpp @@ -256,7 +256,7 @@ class view_serializer } else { - write_float(float_value(m_doc.str(n), n)); + write_float(float_value(m_doc, n)); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) diff --git a/include/nlohmann/detail/view/value.hpp b/include/nlohmann/detail/view/value.hpp index b43a9c4df..23d291634 100644 --- a/include/nlohmann/detail/view/value.hpp +++ b/include/nlohmann/detail/view/value.hpp @@ -49,7 +49,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE T arithmetic_value(const document_data& d, const nod case value_t::number_integer: return static_cast(static_cast(static_cast(integer_bits(n)))); case value_t::number_float: - return static_cast(float_value(d.str(n), n)); + return static_cast(float_value(d, n)); case value_t::boolean: return static_cast((n.flags & node_flags::is_true) != 0); case value_t::null: diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp index 5b67bf9ad..c658e3352 100644 --- a/single_include/nlohmann/json_view.hpp +++ b/single_include/nlohmann/json_view.hpp @@ -2537,13 +2537,19 @@ NLOHMANN_JSON_NAMESPACE_END #include // size_t +#include // int64_t, uint64_t #include // string +#include // integral_constant // #include +// #include + // #include // #include +// #include + NLOHMANN_JSON_NAMESPACE_BEGIN namespace detail @@ -2592,6 +2598,96 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) return convert_float(first, last, dot, mantissa_end); } +/*! +@brief the digits of a float token with at most 19 digits, from its layout + +The digit layout recorded while parsing says where the integer digits, the +fraction digits, and the exponent are, so the digits are read eight at a +time without scanning. + +@param[in] p first character of the token +@param[in] e end of the token +@param[in] limit end of the readable memory (the source text) +*/ +NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + const bool negative = *p == '-'; + p += negative ? 1 : 0; + std::uint64_t w = parse_upto19(p, int_digits, limit); + p += int_digits; + std::int64_t q = 0; + if (frac_digits != 0) + { + w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit); + p += 1 + frac_digits; + q = -static_cast(frac_digits); + } + if (p != e) + { + // [eE][+-]digits; huge exponents saturate (the parser rejected overflow) + ++p; + const bool exp_negative = *p == '-'; + p += (*p == '-' || *p == '+') ? 1 : 0; + std::int64_t exp_value = 0; + for (; p != e; ++p) + { + if (exp_value < 0x10000000) + { + exp_value = (exp_value * 10) + (*p - '0'); + } + } + q += exp_negative ? -exp_value : exp_value; + } + + float_significand d; + d.w = w; + d.exponent = q; + d.negative = negative; + return d; +} + +/*! +@brief the value of a float token with at most 19 digits, from its layout + +The result is correctly rounded by the lexer's conversion +(detail::decimal_to_float(): Clinger's fast path where both operands are +exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19 +digits), so it is the value parse() produces. +*/ +template +NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept +{ + return decimal_to_float(layout_decimal(p, e, int_digits, frac_digits, limit)); +} + +/// the value of the float token of a node, as parse() converts it; floats and +/// doubles with at most 19 digits are converted from the digit layout +template +FloatType float_value(const document_data& d, const node& n) +{ + return float_value(d, n, std::integral_constant::value> {}); +} + +template +FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/) +{ + const unsigned int_digits = n.extra & 0xFFu; + const unsigned frac_digits = n.extra >> 8u; + if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many") + { + // (a float token not written by an edit is in the text) + const auto* const first = reinterpret_cast(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return layout_float(first, first + n.len, int_digits, frac_digits, reinterpret_cast(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + return float_value(d.str(n), n); +} + +template +FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/) +{ + return float_value(d.str(n), n); +} + } // namespace view } // namespace detail NLOHMANN_JSON_NAMESPACE_END @@ -2660,7 +2756,7 @@ BasicJsonType materialize(const document_data& d, const node* n) ++n; break; case value_t::number_float: - sax.number_float(float_value(d.str(*n), *n), no_token); + sax.number_float(float_value(d, *n), no_token); ++n; break; case value_t::boolean: @@ -3149,7 +3245,7 @@ class view_serializer } else { - write_float(float_value(m_doc.str(n), n)); + write_float(float_value(m_doc, n)); } break; case value_t::object: // LCOV_EXCL_LINE (containers are written by dump()) @@ -3482,7 +3578,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE T arithmetic_value(const document_data& d, const nod case value_t::number_integer: return static_cast(static_cast(static_cast(integer_bits(n)))); case value_t::number_float: - return static_cast(float_value(d.str(n), n)); + return static_cast(float_value(d, n)); case value_t::boolean: return static_cast((n.flags & node_flags::is_true) != 0); case value_t::null: diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp index e61f2d5d5..6729df7e6 100644 --- a/tests/src/unit-json_view.cpp +++ b/tests/src/unit-json_view.cpp @@ -873,7 +873,15 @@ TEST_CASE("json_view values") std::mt19937_64 rng(5295); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) std::vector tokens = {"0.1", "-0.0", "1e308", "1.7976931348623157e308", "2.2250738585072011e-308", "4.9e-324", "5e-324", "0.1000000000000000055511151231257827021181583404541015625", "123456789012345678901234567890", - "9007199254740993", "1.00000000000000011102230246251565404236316680908203125", "7.2057594037927933e16" + "9007199254740993", "1.00000000000000011102230246251565404236316680908203125", "7.2057594037927933e16", + // around the limits of the conversion from the digit layout: 19 and 20 + // digits, and those of Clinger's fast path (2^53, 10^22) + "1234567890.123456789", "1234567890.1234567891", "0.0000000000000000001", "123456789012345678.9", + "9007199254740992.0", "9007199254740993.0", "9007199254740994.0", "1.5e22", "1.5e23", "15e-22", "15e-23", + "1e-400", "0.0e0", "-0.0e-5", "12E+3", "12e-0", + // and of float: 2^24 + 1 and 2^24 + 3 (ties), the subnormal and normal limits + "16777217", "16777219", "1.4e-45", "7.006492321624085e-46", "7.006492321624086e-46", + "1.17549435e-38", "0.30000001192092896", "3.4028234e37" }; for (int i = 0; i < 20000; ++i) {