Convert the floats of json_view from the digit layout

The parser records where the integer digits, the fraction digits, and the
exponent of a float token are. For floats and doubles with at most 19
digits, the value is now read from that layout: the digits eight at a time,
without scanning the token, and rounded by the library's conversion core
(detail::decimal_to_float(): Clinger's fast path where both operands are
exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19
digits). It rounds correctly, so the values are those of parse(); other
tokens and types keep the library's conversion of the whole token.

get<double>(), materialize(), dump(), and comparisons use it. Traversing
canada.json (111,000 floats, every number converted): 0.95 -> 1.29 GB/s.

Tests add tokens around the limits (19 and 20 digits, 2^53, 10^22, and
those of float) to the bit-for-bit comparison with parse().

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-28 23:24:57 +02:00
parent 4f00cac911
commit cbd93ff4bc
6 changed files with 205 additions and 7 deletions

View File

@@ -2537,13 +2537,19 @@ NLOHMANN_JSON_NAMESPACE_END
#include <cstddef> // size_t
#include <cstdint> // int64_t, uint64_t
#include <string> // string
#include <type_traits> // integral_constant
// #include <nlohmann/json.hpp>
// #include <nlohmann/detail/view/document_data.hpp>
// #include <nlohmann/detail/view/macro_scope.hpp>
// #include <nlohmann/detail/view/node.hpp>
// #include <nlohmann/detail/view/scan.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -2592,6 +2598,96 @@ NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n)
return convert_float<FloatType>(first, last, dot, mantissa_end);
}
/*!
@brief the digits of a float token with at most 19 digits, from its layout
The digit layout recorded while parsing says where the integer digits, the
fraction digits, and the exponent are, so the digits are read eight at a
time without scanning.
@param[in] p first character of the token
@param[in] e end of the token
@param[in] limit end of the readable memory (the source text)
*/
NLOHMANN_VIEW_ALWAYS_INLINE float_significand layout_decimal(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
{
const bool negative = *p == '-';
p += negative ? 1 : 0;
std::uint64_t w = parse_upto19(p, int_digits, limit);
p += int_digits;
std::int64_t q = 0;
if (frac_digits != 0)
{
w = (w * int_pow10(frac_digits)) + parse_upto19(p + 1, frac_digits, limit);
p += 1 + frac_digits;
q = -static_cast<std::int64_t>(frac_digits);
}
if (p != e)
{
// [eE][+-]digits; huge exponents saturate (the parser rejected overflow)
++p;
const bool exp_negative = *p == '-';
p += (*p == '-' || *p == '+') ? 1 : 0;
std::int64_t exp_value = 0;
for (; p != e; ++p)
{
if (exp_value < 0x10000000)
{
exp_value = (exp_value * 10) + (*p - '0');
}
}
q += exp_negative ? -exp_value : exp_value;
}
float_significand d;
d.w = w;
d.exponent = q;
d.negative = negative;
return d;
}
/*!
@brief the value of a float token with at most 19 digits, from its layout
The result is correctly rounded by the lexer's conversion
(detail::decimal_to_float(): Clinger's fast path where both operands are
exact, else the Eisel-Lemire algorithm, which needs no fallback for up to 19
digits), so it is the value parse() produces.
*/
template<typename FloatType>
NLOHMANN_VIEW_ALWAYS_INLINE FloatType layout_float(const unsigned char* p, const unsigned char* e, unsigned int_digits, unsigned frac_digits, const unsigned char* limit) noexcept
{
return decimal_to_float<FloatType>(layout_decimal(p, e, int_digits, frac_digits, limit));
}
/// the value of the float token of a node, as parse() converts it; floats and
/// doubles with at most 19 digits are converted from the digit layout
template<typename FloatType>
FloatType float_value(const document_data& d, const node& n)
{
return float_value<FloatType>(d, n, std::integral_constant<bool, has_native_float_format<FloatType>::value> {});
}
template<typename FloatType>
FloatType float_value(const document_data& d, const node& n, std::true_type /*binary32 or binary64*/)
{
const unsigned int_digits = n.extra & 0xFFu;
const unsigned frac_digits = n.extra >> 8u;
if (NLOHMANN_VIEW_LIKELY(int_digits + frac_digits <= 19)) // (255 marks "many")
{
// (a float token not written by an edit is in the text)
const auto* const first = reinterpret_cast<const unsigned char*>(d.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
return layout_float<FloatType>(first, first + n.len, int_digits, frac_digits, reinterpret_cast<const unsigned char*>(d.src + d.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
}
return float_value<FloatType>(d.str(n), n);
}
template<typename FloatType>
FloatType float_value(const document_data& d, const node& n, std::false_type /*other*/)
{
return float_value<FloatType>(d.str(n), n);
}
} // namespace view
} // namespace detail
NLOHMANN_JSON_NAMESPACE_END
@@ -2660,7 +2756,7 @@ BasicJsonType materialize(const document_data& d, const node* n)
++n;
break;
case value_t::number_float:
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d.str(*n), *n), no_token);
sax.number_float(float_value<typename BasicJsonType::number_float_t>(d, *n), no_token);
++n;
break;
case value_t::boolean:
@@ -3149,7 +3245,7 @@ class view_serializer
}
else
{
write_float(float_value<number_float_t>(m_doc.str(n), n));
write_float(float_value<number_float_t>(m_doc, n));
}
break;
case value_t::object: // LCOV_EXCL_LINE (containers are written by dump())
@@ -3482,7 +3578,7 @@ NLOHMANN_VIEW_ALWAYS_INLINE T arithmetic_value(const document_data& d, const nod
case value_t::number_integer:
return static_cast<T>(static_cast<typename BasicJsonType::number_integer_t>(static_cast<std::int64_t>(integer_bits(n))));
case value_t::number_float:
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d.str(n), n));
return static_cast<T>(float_value<typename BasicJsonType::number_float_t>(d, n));
case value_t::boolean:
return static_cast<T>((n.flags & node_flags::is_true) != 0);
case value_t::null: