Write the floats of json_view from their digits

dump() writes a float token of at most 15 significant digits from its
digits, without converting it to a double and back: two decimals of at
most 15 digits are farther apart than the rounding interval of a
normal double (the argument behind DBL_DIG), so the token's digits are
the shortest ones of its double, which the library's conversion (Zmij)
writes. The exponent must keep the value away from subnormals and
overflow. Longer tokens are converted from the digits already read.

Doubles are written into the output directly instead of through a
local buffer. With NEON, the fixed layouts ("12.5", "0.001", "100.0")
are put together in vector registers by a table lookup of the digit
bytes: the portable layout copies the digits through a buffer at
another offset, and a load that spans several recent stores waits
until they reach the cache.

dump() of float-heavy documents: numbers -69%, marine_ik -62%,
mesh.pretty -34%, canada (mostly 16 or 17 digits) -14%.

Tests: 20,000 float tokens of 1 to 17 significant digits in every
spelling (point, exponent, leading and trailing zeros, sign), from about
1e-320 to 1e300, written as json::dump() writes them. On AArch64 they
check the NEON layout; x86 and JSON_VIEW_NO_SIMD use the library's.
Other float types, now the only ones on the general path, are tested
with non-finite values set by edits (written as null).

Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
Niels Lohmann
2026-09-29 03:33:19 +02:00
parent 3336423f04
commit ebcc734e45
4 changed files with 352 additions and 4 deletions

View File

@@ -23,6 +23,7 @@
#include <nlohmann/detail/view/macro_scope.hpp>
#include <nlohmann/detail/view/node.hpp>
#include <nlohmann/detail/view/number.hpp>
#include <nlohmann/detail/view/simd.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -221,6 +222,69 @@ struct dump_style
bool source_numbers = false; ///< copy number tokens from the source
};
/*!
@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0,
at most 17 digits; up to 41 bytes are written at first)
With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in
vector registers: output byte i is byte s + i of the digits (after '0's)
before the point, and byte s + i - 1 after it. The portable code writes the
digits to a buffer and copies them from there at another offset, and a load
that spans several recent stores waits until they reach the cache.
*/
NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
{
#if NLOHMANN_VIEW_NEON
namespace dtoa = ::nlohmann::detail::dtoa_impl;
const std::uint64_t upper = digits / 100000000u;
const std::uint64_t b0 = upper / 100000000u; // one digit
const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u);
const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u);
// leading and trailing zero digits (as dtoa_impl::write_decimal())
int leading = 7;
if (b0 == 0)
{
leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8);
}
int zeros = 16;
if (b2 != 0)
{
zeros = count_trailing_zeros(b2) / 8;
}
else if (b1 != 0)
{
zeros = 8 + (count_trailing_zeros(b1) / 8);
}
const int k = 24 - leading - zeros; // significant digits
const int n = k + exp + zeros; // position of the point after the first digit
if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15))
{
static const std::array<std::uint8_t, 32> iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}};
const int pad = n <= 0 ? 1 - n : 0;
const int len = k + pad;
const int point = n + pad;
// the 24 digit bytes in memory order, then '0's
const std::uint64_t zero_chars = 0x3030303030303030u;
const uint8x16x2_t table = {{
vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))),
vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0'))
}
};
const uint8x16_t s = vdupq_n_u8(static_cast<std::uint8_t>(leading - pad));
const uint8x16_t at_point = vdupq_n_u8(static_cast<std::uint8_t>(point));
for (std::size_t half = 0; half < 2; ++half)
{
const uint8x16_t i = vld1q_u8(iota.data() + (16 * half));
// (+ 0xFF is - 1 after the point; indexes past the digits read a '0')
const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31));
vst1q_u8(reinterpret_cast<std::uint8_t*>(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
}
return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0"
}
#endif
return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp);
}
/*!
@brief write a view's subtree as basic_json::dump() writes the value
@@ -496,10 +560,15 @@ class view_serializer
room(n->len);
copy(src + n->off, n->len);
}
else if (std::is_same<number_float_t, double>::value)
{
room(64);
w = write_double_at(w, *n);
}
else
{
m_out.set_cursor(w);
write_float(float_value<number_float_t>(m_doc, *n));
write_float_node(*n);
w = m_out.cursor();
lim = m_out.limit();
}
@@ -666,7 +735,7 @@ class view_serializer
}
else
{
write_float(float_value<number_float_t>(m_doc, n));
write_float_node(n);
}
break;
case value_t::object: // LCOV_EXCL_LINE (containers are written by dump())
@@ -678,6 +747,83 @@ class view_serializer
}
}
/// a float node as dump() writes it
void write_float_node(const node& n)
{
write_float_node(n, std::is_same<number_float_t, double> {});
}
void write_float_node(const node& n, std::false_type /*other*/)
{
write_float(float_value<number_float_t>(m_doc, n));
}
void write_float_node(const node& n, std::true_type /*double*/)
{
m_out.reserve(64);
m_out.set_cursor(write_double_at(m_out.cursor(), n));
}
/*!
@brief (doubles) the float at n as dump() writes it, at w (64 bytes of room)
A token of at most 15 significant digits is written from its digits,
without a conversion: two decimals of at most 15 digits are farther
apart than the rounding interval of a (normal) double (the argument
behind DBL_DIG), so the token's digits are the shortest ones of its
double, which the library's conversion writes (Zmij). Other tokens are
converted from the digits already read.
*/
char* write_double_at(char* w, const node& n)
{
const unsigned int_digits = n.extra & 0xFFu;
const unsigned frac_digits = n.extra >> 8u;
if ((n.flags & node_flags::storage) != node_flags::edited && int_digits + frac_digits <= 19)
{
const auto* const first = reinterpret_cast<const unsigned char*>(m_doc.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
const float_significand d = layout_decimal(first, first + n.len, int_digits, frac_digits, reinterpret_cast<const unsigned char*>(m_doc.src + m_doc.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
// (the exponent keeps the value far from subnormals and overflow)
if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290)
{
*w = '-';
return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast<int>(d.exponent));
}
return write_double_value_at(w, decimal_to_float<double>(d)); // (without reading the token again)
}
return write_double_value_at(w, static_cast<double>(float_value<number_float_t>(m_doc, n)));
}
/// n bytes of text at w
static char* write_text_at(char* w, const char* text, std::size_t n) noexcept
{
std::memcpy(w, text, n);
return w + n;
}
/// a double as dump() writes it, at w (64 bytes of room)
static char* write_double_value_at(char* w, double x)
{
if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x)))
{
return write_text_at(w, "null", 4);
}
#if NLOHMANN_VIEW_NEON
std::uint64_t bits = 0;
std::memcpy(&bits, &x, sizeof(bits));
*w = '-';
w += bits >> 63u;
bits &= ~(std::uint64_t{1} << 63u);
if (bits == 0)
{
return write_text_at(w, "0.0", 3);
}
const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits);
return write_decimal(w, d.significand, d.exponent);
#else
return ::nlohmann::detail::to_chars(w, w + 64, x);
#endif
}
/// as serializer::dump_float()
void write_float(number_float_t x)
{

View File

@@ -5323,6 +5323,8 @@ NLOHMANN_JSON_NAMESPACE_END
// #include <nlohmann/detail/view/number.hpp>
// #include <nlohmann/detail/view/simd.hpp>
NLOHMANN_JSON_NAMESPACE_BEGIN
namespace detail
@@ -5521,6 +5523,69 @@ struct dump_style
bool source_numbers = false; ///< copy number tokens from the source
};
/*!
@brief digits * 10^exp as dtoa_impl::write_decimal() writes it (digits not 0,
at most 17 digits; up to 41 bytes are written at first)
With NEON, the fixed layouts ("0.00123", "12.5", "100.0") are put together in
vector registers: output byte i is byte s + i of the digits (after '0's)
before the point, and byte s + i - 1 after it. The portable code writes the
digits to a buffer and copies them from there at another offset, and a load
that spans several recent stores waits until they reach the cache.
*/
NLOHMANN_VIEW_ALWAYS_INLINE char* write_decimal(char* first, std::uint64_t digits, int exp) noexcept
{
#if NLOHMANN_VIEW_NEON
namespace dtoa = ::nlohmann::detail::dtoa_impl;
const std::uint64_t upper = digits / 100000000u;
const std::uint64_t b0 = upper / 100000000u; // one digit
const std::uint64_t b1 = dtoa::eight_digit_bytes(upper % 100000000u);
const std::uint64_t b2 = dtoa::eight_digit_bytes(digits % 100000000u);
// leading and trailing zero digits (as dtoa_impl::write_decimal())
int leading = 7;
if (b0 == 0)
{
leading = b1 != 0 ? 8 + (count_leading_zeros(b1) / 8) : 16 + (count_leading_zeros(b2) / 8);
}
int zeros = 16;
if (b2 != 0)
{
zeros = count_trailing_zeros(b2) / 8;
}
else if (b1 != 0)
{
zeros = 8 + (count_trailing_zeros(b1) / 8);
}
const int k = 24 - leading - zeros; // significant digits
const int n = k + exp + zeros; // position of the point after the first digit
if (NLOHMANN_VIEW_LIKELY(-4 < n && n <= 15))
{
static const std::array<std::uint8_t, 32> iota = {{0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 26, 27, 28, 29, 30, 31}};
const int pad = n <= 0 ? 1 - n : 0;
const int len = k + pad;
const int point = n + pad;
// the 24 digit bytes in memory order, then '0's
const std::uint64_t zero_chars = 0x3030303030303030u;
const uint8x16x2_t table = {{
vcombine_u8(vcreate_u8(__builtin_bswap64(b0 + zero_chars)), vcreate_u8(__builtin_bswap64(b1 + zero_chars))),
vcombine_u8(vcreate_u8(__builtin_bswap64(b2 + zero_chars)), vdup_n_u8('0'))
}
};
const uint8x16_t s = vdupq_n_u8(static_cast<std::uint8_t>(leading - pad));
const uint8x16_t at_point = vdupq_n_u8(static_cast<std::uint8_t>(point));
for (std::size_t half = 0; half < 2; ++half)
{
const uint8x16_t i = vld1q_u8(iota.data() + (16 * half));
// (+ 0xFF is - 1 after the point; indexes past the digits read a '0')
const uint8x16_t index = vminq_u8(vaddq_u8(vaddq_u8(i, s), vcgtq_u8(i, at_point)), vdupq_n_u8(31));
vst1q_u8(reinterpret_cast<std::uint8_t*>(first) + (16 * half), vbslq_u8(vceqq_u8(i, at_point), vdupq_n_u8('.'), vqtbl2q_u8(table, index))); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
}
return first + (point >= len ? point + 2 : len + 1); // "digits[000].0" ends after ".0"
}
#endif
return ::nlohmann::detail::dtoa_impl::write_decimal(first, digits, exp);
}
/*!
@brief write a view's subtree as basic_json::dump() writes the value
@@ -5796,10 +5861,15 @@ class view_serializer
room(n->len);
copy(src + n->off, n->len);
}
else if (std::is_same<number_float_t, double>::value)
{
room(64);
w = write_double_at(w, *n);
}
else
{
m_out.set_cursor(w);
write_float(float_value<number_float_t>(m_doc, *n));
write_float_node(*n);
w = m_out.cursor();
lim = m_out.limit();
}
@@ -5966,7 +6036,7 @@ class view_serializer
}
else
{
write_float(float_value<number_float_t>(m_doc, n));
write_float_node(n);
}
break;
case value_t::object: // LCOV_EXCL_LINE (containers are written by dump())
@@ -5978,6 +6048,83 @@ class view_serializer
}
}
/// a float node as dump() writes it
void write_float_node(const node& n)
{
write_float_node(n, std::is_same<number_float_t, double> {});
}
void write_float_node(const node& n, std::false_type /*other*/)
{
write_float(float_value<number_float_t>(m_doc, n));
}
void write_float_node(const node& n, std::true_type /*double*/)
{
m_out.reserve(64);
m_out.set_cursor(write_double_at(m_out.cursor(), n));
}
/*!
@brief (doubles) the float at n as dump() writes it, at w (64 bytes of room)
A token of at most 15 significant digits is written from its digits,
without a conversion: two decimals of at most 15 digits are farther
apart than the rounding interval of a (normal) double (the argument
behind DBL_DIG), so the token's digits are the shortest ones of its
double, which the library's conversion writes (Zmij). Other tokens are
converted from the digits already read.
*/
char* write_double_at(char* w, const node& n)
{
const unsigned int_digits = n.extra & 0xFFu;
const unsigned frac_digits = n.extra >> 8u;
if ((n.flags & node_flags::storage) != node_flags::edited && int_digits + frac_digits <= 19)
{
const auto* const first = reinterpret_cast<const unsigned char*>(m_doc.src + n.off); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
const float_significand d = layout_decimal(first, first + n.len, int_digits, frac_digits, reinterpret_cast<const unsigned char*>(m_doc.src + m_doc.size)); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
// (the exponent keeps the value far from subnormals and overflow)
if (d.w != 0 && d.w < 1000000000000000u && d.exponent >= -290 && d.exponent <= 290)
{
*w = '-';
return write_decimal(w + (d.negative ? 1 : 0), d.w, static_cast<int>(d.exponent));
}
return write_double_value_at(w, decimal_to_float<double>(d)); // (without reading the token again)
}
return write_double_value_at(w, static_cast<double>(float_value<number_float_t>(m_doc, n)));
}
/// n bytes of text at w
static char* write_text_at(char* w, const char* text, std::size_t n) noexcept
{
std::memcpy(w, text, n);
return w + n;
}
/// a double as dump() writes it, at w (64 bytes of room)
static char* write_double_value_at(char* w, double x)
{
if (NLOHMANN_VIEW_UNLIKELY(!std::isfinite(x)))
{
return write_text_at(w, "null", 4);
}
#if NLOHMANN_VIEW_NEON
std::uint64_t bits = 0;
std::memcpy(&bits, &x, sizeof(bits));
*w = '-';
w += bits >> 63u;
bits &= ~(std::uint64_t{1} << 63u);
if (bits == 0)
{
return write_text_at(w, "0.0", 3);
}
const ::nlohmann::detail::zmij::decimal d = ::nlohmann::detail::zmij::to_decimal(bits);
return write_decimal(w, d.significand, d.exponent);
#else
return ::nlohmann::detail::to_chars(w, w + 64, x);
#endif
}
/// as serializer::dump_float()
void write_float(number_float_t x)
{

View File

@@ -1151,6 +1151,47 @@ TEST_CASE("json_view dump")
CHECK(d.root().dump(0, ' ', false, json_view::number_format::source) == "[\n1.50,\n1E2,\n-0,\n-0.0,\n123456789012345678901234567890,\n18446744073709551615,\n-9223372036854775808,\n0.1,\n1e-7,\n5e-324\n]");
CHECK(d.root().dump(-1, ' ', true, json_view::number_format::source) == "[1.50,1E2,-0,-0.0,123456789012345678901234567890,18446744073709551615,-9223372036854775808,0.1,1e-7,5e-324]");
// float tokens of up to 17 significant digits in every spelling: those
// of at most 15 digits are written from their digits, the others
// through the conversion; both as dump() writes them
{
std::mt19937_64 tokens(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed)
std::string many_tokens = "[";
for (int i = 0; i < 20000; ++i)
{
const auto length = static_cast<std::size_t>(1 + (tokens() % 17));
std::string digits(1, static_cast<char>('1' + (tokens() % 9)));
for (std::size_t k = 1; k < length; ++k)
{
digits += static_cast<char>('0' + (tokens() % 10));
}
digits += std::string(tokens() % 4, '0'); // trailing zeros
std::string token = tokens() % 3 == 0 ? "-" : "";
const auto point = static_cast<std::size_t>(tokens() % (digits.size() + 1));
if (point == 0)
{
token += "0." + std::string(tokens() % 5, '0') + digits;
}
else
{
token += digits.substr(0, point) + (point < digits.size() ? "." + digits.substr(point) : "");
}
// an exponent that keeps the value between about 1e-320 and 1e300
const int exponent = static_cast<int>(tokens() % 600) - 300 - static_cast<int>(point);
if (tokens() % 4 != 0)
{
token += (tokens() % 2 == 0 ? "e" : "E") + std::string(exponent >= 0 && tokens() % 2 == 0 ? "+" : "") + std::to_string(exponent);
}
else if (point == digits.size())
{
token += ".0"; // (a float, not an integer)
}
many_tokens += (i != 0 ? "," : "") + token;
}
many_tokens += ']';
CHECK(json_document::parse(many_tokens).root().dump() == json::parse(many_tokens).dump());
}
// random doubles, written as parse() and dump() would
std::mt19937_64 rng(1170); // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed)
std::string many = "[";

View File

@@ -26,6 +26,7 @@ using ptr_t = ordered_json::json_pointer;
#include <functional>
#include <iterator>
#include <limits>
#include <map>
#include <random>
#include <string>
#include <vector>
@@ -481,6 +482,19 @@ TEST_CASE("json_view edits: views and values")
CHECK(d.root().materialize().dump() == json::parse(R"([1.5, 100.0, 0.1, null, null, 18446744073709551615, -9223372036854775808])").dump());
}
SECTION("numbers of other float types")
{
// doubles have their own path to the output; other float types are
// written as basic_json writes them, non-finite values as null
using json_float = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
using document_float = nlohmann::basic_json_document<json_float, true>;
document_float d = document_float::parse("[1.5]");
d.push_back(d.root(), std::numeric_limits<float>::quiet_NaN());
d.push_back(d.root(), -std::numeric_limits<float>::infinity());
CHECK(d.root().dump() == "[1.5,null,null]");
CHECK(d.root().dump(2) == json_float::parse("[1.5, null, null]").dump(2));
}
SECTION("nulls become containers, and the root can be replaced")
{
json_editable_document d = json_editable_document::parse("[null, null]");