From b98aef8a0708244feb8b033d1702be6a3d9f7cf1 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 29 Sep 2026 23:52:34 +0200 Subject: [PATCH] Round-trip BJData ND-array annotations exactly (single precision, key order) to_bjdata() encoded a JData-annotated object as a BJData ND-array in two cases where from_bjdata() then returned a different value, breaking the documented round-trip guarantee: 1. A "single" element that is finite and in range but not exactly representable as float (e.g. 0.1) or that underflows to 0 (e.g. 1e-300) was silently narrowed instead of falling back to a plain object, unlike out-of-range integer elements. write_bjdata_ndarray() now only accepts a "single" element if it survives the narrowing to float and back, the same criterion write_compact_float() already uses for CBOR/MessagePack. 2. from_bjdata() emitted the annotation keys as _ArraySize_, _ArrayType_, _ArrayData_ instead of the documented _ArrayType_, _ArraySize_, _ArrayData_, because the size key is written while the dimension vector is read, before the type key. For ordered_json, whose comparison takes key order into account, this made a round trip of the documented example compare unequal. The element type marker is known before the dimension vector is read (it precedes '#'), so it is now passed down and the "_ArrayType_" key is emitted first. Fixes #5661. Signed-off-by: Niels Lohmann --- .../docs/features/binary_formats/bjdata.md | 10 ++- .../nlohmann/detail/input/binary_reader.hpp | 63 ++++++++++----- .../nlohmann/detail/output/binary_writer.hpp | 16 +++- single_include/nlohmann/json.hpp | 79 +++++++++++++------ tests/src/unit-bjdata.cpp | 46 ++++++++++- 5 files changed, 162 insertions(+), 52 deletions(-) diff --git a/docs/mkdocs/docs/features/binary_formats/bjdata.md b/docs/mkdocs/docs/features/binary_formats/bjdata.md index a0c84edaf..f3cc2d838 100644 --- a/docs/mkdocs/docs/features/binary_formats/bjdata.md +++ b/docs/mkdocs/docs/features/binary_formats/bjdata.md @@ -132,8 +132,14 @@ The library uses the following mapping from JSON values types to BJData types ac parsed back as a regular array, - every entry of `"_ArraySize_"` is a positive integer, and their product is representable as a `std::size_t`, - `"_ArrayData_"` is an array holding exactly that many elements, and - - every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"` (a floating-point number for - `single` and `double`, an integer otherwise). + - every element of `"_ArrayData_"` is a number of the kind named by `"_ArrayType_"`: for the integer types, a + value that fits the named width; for `double`, any value; for `single`, a value that survives narrowing to + `float` and back without change (for instance, `0.1` does not, since it is not exactly representable as + `float`). + + An annotated object is always read back with its keys in the order shown above, `"_ArrayType_"`, `"_ArraySize_"`, + `"_ArrayData_"`, regardless of the order the ND-array's header stores them in on the wire. This matters for + `ordered_json`, whose comparison takes key order into account. The current version of this library does not yet support automatic detection of and conversion from a nested JSON array input to a BJData ND-array. diff --git a/include/nlohmann/detail/input/binary_reader.hpp b/include/nlohmann/detail/input/binary_reader.hpp index b9e6b304b..587d98e64 100644 --- a/include/nlohmann/detail/input/binary_reader.hpp +++ b/include/nlohmann/detail/input/binary_reader.hpp @@ -2728,10 +2728,15 @@ class binary_reader is_ndarray can only return `true` when its initial value is `false` @param[in] prefix type marker if already read, otherwise set to 0 + @param[in] ndarray_dtype the element type marker of the enclosing bjdata ndarray if + already known (it precedes the dimension vector read here), + otherwise 0; used to emit the "_ArrayType_" annotation key + before "_ArraySize_" if a dimension vector turns out to + describe an ndarray @return whether size determination completed */ - bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0) + bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0, char_int_type ndarray_dtype = 0) { if (prefix == 0) { @@ -2901,8 +2906,37 @@ class binary_reader } } + if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3))) + { + return false; + } + + // the element type precedes the dimension vector (see get_ubjson_size_type) + // and is passed down as ndarray_dtype; emit it here so the annotation keys + // follow the documented _ArrayType_, _ArraySize_, _ArrayData_ order + if (ndarray_dtype != 0) + { + auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), ndarray_dtype, [](const bjd_type & p, char_int_type t) + { + return p.first < t; + }); + if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != ndarray_dtype)) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr)); + } + + string_t type_key = "_ArrayType_"; + string_t type = it->second; // sax->string() takes a reference + if (JSON_HEDLEY_UNLIKELY(!sax->key(type_key) || !sax->string(type))) + { + return false; + } + } + string_t key = "_ArraySize_"; - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3) || !sax->key(key) || !sax->start_array(dim.size()))) + if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(dim.size()))) { return false; } @@ -3003,7 +3037,7 @@ class binary_reader exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr)); } - const bool is_error = get_ubjson_size_value(result.first, is_ndarray); + const bool is_error = get_ubjson_size_value(result.first, is_ndarray, 0, result.second); // an ndarray was read here only if the flag flipped; when it was // seeded true, get_ubjson_size_value() already rejected the nested // dimension vector @@ -3239,30 +3273,17 @@ class binary_reader if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) { size_and_type.second &= ~(static_cast(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker - auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t) - { - return p.first < t; - }); - string_t key = "_ArrayType_"; - if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second)) - { - auto last_token = get_token_string(); - return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr)); - } - - string_t type = it->second; // sax->string() takes a reference - if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type))) - { - return false; - } + // the "_ArrayType_" and "_ArraySize_" annotation keys were already emitted by + // get_ubjson_size_value() (the type marker is known before the dimension vector + // that determines size_and_type.first is read, so it is emitted first there to + // match the documented _ArrayType_, _ArraySize_, _ArrayData_ key order) if (size_and_type.second == 'C' || size_and_type.second == 'B') { size_and_type.second = 'U'; } - key = "_ArrayData_"; + string_t key = "_ArrayData_"; if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) )) { return false; diff --git a/include/nlohmann/detail/output/binary_writer.hpp b/include/nlohmann/detail/output/binary_writer.hpp index 9da4269a8..c486496fb 100644 --- a/include/nlohmann/detail/output/binary_writer.hpp +++ b/include/nlohmann/detail/output/binary_writer.hpp @@ -1991,9 +1991,21 @@ class binary_writer case 'd': { const auto dval = el.template get(); - in_range = !std::isfinite(dval) || +#ifdef __GNUC__ + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") +#endif + // a value that would be rounded (rather than exactly represented) by the + // narrowing to float is treated like an out-of-range integer element above; + // this is the same criterion write_compact_float() uses for CBOR/MessagePack + in_range = std::isnan(dval) || (dval >= static_cast(std::numeric_limits::lowest()) && - dval <= static_cast((std::numeric_limits::max)())); + dval <= static_cast((std::numeric_limits::max)()) && + static_cast(static_cast(dval)) == dval) || + std::isinf(dval); +#ifdef __GNUC__ + JSON_HEDLEY_DIAGNOSTIC_POP +#endif break; } default: diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index 576498738..6138ea7b6 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -15495,10 +15495,15 @@ class binary_reader is_ndarray can only return `true` when its initial value is `false` @param[in] prefix type marker if already read, otherwise set to 0 + @param[in] ndarray_dtype the element type marker of the enclosing bjdata ndarray if + already known (it precedes the dimension vector read here), + otherwise 0; used to emit the "_ArrayType_" annotation key + before "_ArraySize_" if a dimension vector turns out to + describe an ndarray @return whether size determination completed */ - bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0) + bool get_ubjson_size_value(std::size_t& result, bool& is_ndarray, char_int_type prefix = 0, char_int_type ndarray_dtype = 0) { if (prefix == 0) { @@ -15668,8 +15673,37 @@ class binary_reader } } + if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3))) + { + return false; + } + + // the element type precedes the dimension vector (see get_ubjson_size_type) + // and is passed down as ndarray_dtype; emit it here so the annotation keys + // follow the documented _ArrayType_, _ArraySize_, _ArrayData_ order + if (ndarray_dtype != 0) + { + auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), ndarray_dtype, [](const bjd_type & p, char_int_type t) + { + return p.first < t; + }); + if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != ndarray_dtype)) + { + auto last_token = get_token_string(); + return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, + exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr)); + } + + string_t type_key = "_ArrayType_"; + string_t type = it->second; // sax->string() takes a reference + if (JSON_HEDLEY_UNLIKELY(!sax->key(type_key) || !sax->string(type))) + { + return false; + } + } + string_t key = "_ArraySize_"; - if (JSON_HEDLEY_UNLIKELY(!sax->start_object(3) || !sax->key(key) || !sax->start_array(dim.size()))) + if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(dim.size()))) { return false; } @@ -15770,7 +15804,7 @@ class binary_reader exception_message(input_format, concat("expected '#' after type information; last byte: 0x", last_token), "size"), nullptr)); } - const bool is_error = get_ubjson_size_value(result.first, is_ndarray); + const bool is_error = get_ubjson_size_value(result.first, is_ndarray, 0, result.second); // an ndarray was read here only if the flag flipped; when it was // seeded true, get_ubjson_size_value() already rejected the nested // dimension vector @@ -16006,30 +16040,17 @@ class binary_reader if (input_format == input_format_t::bjdata && size_and_type.first != npos && (size_and_type.second & (1 << 8)) != 0) { size_and_type.second &= ~(static_cast(1) << 8); // use bit 8 to indicate ndarray, here we remove the bit to restore the type marker - auto it = std::lower_bound(bjd_types_map.begin(), bjd_types_map.end(), size_and_type.second, [](const bjd_type & p, char_int_type t) - { - return p.first < t; - }); - string_t key = "_ArrayType_"; - if (JSON_HEDLEY_UNLIKELY(it == bjd_types_map.end() || it->first != size_and_type.second)) - { - auto last_token = get_token_string(); - return sax->parse_error(chars_read, last_token, parse_error::create(112, chars_read, - exception_message(input_format, "invalid byte: 0x" + last_token, "type"), nullptr)); - } - - string_t type = it->second; // sax->string() takes a reference - if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->string(type))) - { - return false; - } + // the "_ArrayType_" and "_ArraySize_" annotation keys were already emitted by + // get_ubjson_size_value() (the type marker is known before the dimension vector + // that determines size_and_type.first is read, so it is emitted first there to + // match the documented _ArrayType_, _ArraySize_, _ArrayData_ key order) if (size_and_type.second == 'C' || size_and_type.second == 'B') { size_and_type.second = 'U'; } - key = "_ArrayData_"; + string_t key = "_ArrayData_"; if (JSON_HEDLEY_UNLIKELY(!sax->key(key) || !sax->start_array(size_and_type.first) )) { return false; @@ -22321,9 +22342,21 @@ class binary_writer case 'd': { const auto dval = el.template get(); - in_range = !std::isfinite(dval) || +#ifdef __GNUC__ + JSON_HEDLEY_DIAGNOSTIC_PUSH + JSON_HEDLEY_PRAGMA(GCC diagnostic ignored "-Wfloat-equal") +#endif + // a value that would be rounded (rather than exactly represented) by the + // narrowing to float is treated like an out-of-range integer element above; + // this is the same criterion write_compact_float() uses for CBOR/MessagePack + in_range = std::isnan(dval) || (dval >= static_cast(std::numeric_limits::lowest()) && - dval <= static_cast((std::numeric_limits::max)())); + dval <= static_cast((std::numeric_limits::max)()) && + static_cast(static_cast(dval)) == dval) || + std::isinf(dval); +#ifdef __GNUC__ + JSON_HEDLEY_DIAGNOSTIC_POP +#endif break; } default: diff --git a/tests/src/unit-bjdata.cpp b/tests/src/unit-bjdata.cpp index a58507c15..e7ef51392 100644 --- a/tests/src/unit-bjdata.cpp +++ b/tests/src/unit-bjdata.cpp @@ -11,6 +11,7 @@ #define JSON_TESTS_PRIVATE #include using nlohmann::json; +using ordered_json = nlohmann::ordered_json; #include #include @@ -2294,29 +2295,33 @@ TEST_CASE("BJData") SECTION("start_array() in ndarray _ArraySize_") { + // _ArrayType_ (2 events: key + string) is now emitted before + // _ArraySize_ (see GitHub issue #5661), which shifts the events + // below later by the same 2 events std::vector const v = {'[', '$', 'i', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2}; - SaxCountdown scp(2); + SaxCountdown scp(4); CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata)); } SECTION("number_integer() in ndarray _ArraySize_") { std::vector const v = {'[', '$', 'U', '#', '[', '$', 'i', '#', 'i', 2, 2, 1, 1, 2}; - SaxCountdown scp(3); + SaxCountdown scp(5); CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata)); } SECTION("key() in ndarray _ArrayType_") { + // _ArrayType_ is emitted right after start_object(), before _ArraySize_ std::vector const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4}; - SaxCountdown scp(6); + SaxCountdown scp(1); CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata)); } SECTION("string() in ndarray _ArrayType_") { std::vector const v = {'[', '$', 'U', '#', '[', '$', 'U', '#', 'i', 2, 2, 2, 1, 2, 3, 4}; - SaxCountdown scp(7); + SaxCountdown scp(2); CHECK_FALSE(json::sax_parse(v, &scp, json::input_format_t::bjdata)); } @@ -2919,6 +2924,22 @@ TEST_CASE("BJData") CHECK(out_single.at(0) == '{'); CHECK(json::from_bjdata(out_single) == j_single); + // a double element that is finite and within the range of "single" + // but is not exactly representable as a float, so narrowing it would + // silently round it (0.1 is read back as 0.10000000149011612); this, + // like the overflow case above, falls back to a plain object (see + // GitHub issue #5661) + json const j_single_rounded = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 0.1}}}); + const auto out_single_rounded = json::to_bjdata(j_single_rounded); + CHECK(out_single_rounded.at(0) == '{'); + CHECK(json::from_bjdata(out_single_rounded) == j_single_rounded); + + // a double element that underflows to 0 when narrowed to "single" + json const j_single_underflow = json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5, 1e-300}}}); + const auto out_single_underflow = json::to_bjdata(j_single_underflow); + CHECK(out_single_underflow.at(0) == '{'); + CHECK(json::from_bjdata(out_single_underflow) == j_single_underflow); + // in-range boundary values still use the compact ndarray encoding json const j_uint8_ok = json({{"_ArrayType_", "uint8"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {0, 255}}}); CHECK(json::to_bjdata(j_uint8_ok) == std::vector({'[', '$', 'U', '#', '[', 'i', 2, 'i', 1, ']', 0, 255})); @@ -2932,6 +2953,23 @@ TEST_CASE("BJData") CHECK(json::from_bjdata(out_single_ok) == json({{"_ArrayType_", "single"}, {"_ArraySize_", {2, 1}}, {"_ArrayData_", {1.5f, -1.5f}}})); } + SECTION("ndarray annotation keys are read back in the documented order") + { + // from_bjdata() must emit the annotation object's keys in the order + // used throughout the documentation, _ArrayType_, _ArraySize_, + // _ArrayData_: the type marker precedes the dimension vector on the + // wire (see get_ubjson_size_type()), so it is known, and emitted, + // before _ArraySize_. For a plain json this key order is invisible + // (its comparison ignores it), but for an ordered_json it is not (see + // GitHub issue #5661). + const ordered_json o = ordered_json::parse(R"({"_ArrayType_":"uint8","_ArraySize_":[2,2],"_ArrayData_":[1,2,3,4]})"); + const auto packed = ordered_json::to_bjdata(o); + CHECK(packed.at(0) == '['); + const ordered_json o_back = ordered_json::from_bjdata(packed); + CHECK(o_back == o); + CHECK(o_back.dump() == o.dump()); + } + SECTION("ndarray that would not be read back as an annotated object stays as object") { // the reader only restores an annotated object from an ND-array