mirror of
https://github.com/nlohmann/json.git
synced 2026-10-01 22:45:17 +00:00
get_codepoint() read the four hex digits of a \u escape with four calls to get(), each classified by a chain of range comparisons. For contiguous input, get_codepoint_bulk() now decodes them with one lookup per byte (hex_codepoint() in string_scan.hpp, after yyjson's read_hex_u16): a 256-entry table maps a byte to its value, or 0xFF for anything else, and an invalid digit shows in the OR of the four values. It then skips the four bytes and updates the position counters as four get() calls would. If a digit is invalid or fewer than four bytes are left, it changes nothing and the existing loop runs, so errors are reported with the same message and position as before. json::parse, best of 5 runs in separate processes (M1 Max): the escaped twitter.json (every non-ASCII character as \u) -13.6%, all other files within 0.3%. Tests compare the contiguous and the streaming path (value or exception message) for valid escapes, surrogate pairs, truncated and invalid digits at every position, and 3,000 seeded random escapes. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
1765 lines
86 KiB
C++
1765 lines
86 KiB
C++
// __ _____ _____ _____
|
|
// __| | __| | | | JSON for Modern C++ (supporting code)
|
|
// | | |__ | | | | | | version 3.12.0
|
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
|
//
|
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
#include "doctest_compatibility.h"
|
|
|
|
#define JSON_TESTS_PRIVATE
|
|
#include <nlohmann/json.hpp>
|
|
using nlohmann::json;
|
|
|
|
#include <array> // array
|
|
#include <cstdint> // uint32_t, uint64_t
|
|
#include <cstdio> // snprintf
|
|
#include <cstdlib> // strtod
|
|
#include <cstring> // memcpy
|
|
#include <map> // map
|
|
#include <random> // mt19937
|
|
#include <sstream> // stringstream
|
|
#include <string> // string
|
|
#include <utility> // pair
|
|
#include <vector> // vector
|
|
|
|
#include "float_hard_cases.hpp"
|
|
|
|
namespace
|
|
{
|
|
// shortcut to scan a string literal
|
|
json::lexer::token_type scan_string(const char* s, bool ignore_comments = false);
|
|
json::lexer::token_type scan_string(const char* s, const bool ignore_comments)
|
|
{
|
|
auto ia = nlohmann::detail::input_adapter(s);
|
|
return nlohmann::detail::lexer<json, decltype(ia)>(std::move(ia), ignore_comments).scan(); // NOLINT(hicpp-move-const-arg,performance-move-const-arg)
|
|
}
|
|
} // namespace
|
|
|
|
std::string get_error_message(const char* s, bool ignore_comments = false); // NOLINT(misc-use-internal-linkage)
|
|
std::string get_error_message(const char* s, const bool ignore_comments)
|
|
{
|
|
auto ia = nlohmann::detail::input_adapter(s);
|
|
auto lexer = nlohmann::detail::lexer<json, decltype(ia)>(std::move(ia), ignore_comments); // NOLINT(hicpp-move-const-arg,performance-move-const-arg)
|
|
lexer.scan();
|
|
return lexer.get_error_message();
|
|
}
|
|
|
|
TEST_CASE("lexer class")
|
|
{
|
|
SECTION("scan")
|
|
{
|
|
SECTION("structural characters")
|
|
{
|
|
CHECK((scan_string("[") == json::lexer::token_type::begin_array));
|
|
CHECK((scan_string("]") == json::lexer::token_type::end_array));
|
|
CHECK((scan_string("{") == json::lexer::token_type::begin_object));
|
|
CHECK((scan_string("}") == json::lexer::token_type::end_object));
|
|
CHECK((scan_string(",") == json::lexer::token_type::value_separator));
|
|
CHECK((scan_string(":") == json::lexer::token_type::name_separator));
|
|
}
|
|
|
|
SECTION("literal names")
|
|
{
|
|
CHECK((scan_string("null") == json::lexer::token_type::literal_null));
|
|
CHECK((scan_string("true") == json::lexer::token_type::literal_true));
|
|
CHECK((scan_string("false") == json::lexer::token_type::literal_false));
|
|
}
|
|
|
|
SECTION("numbers")
|
|
{
|
|
CHECK((scan_string("0") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("1") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("2") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("3") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("4") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("5") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("6") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("7") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("8") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("9") == json::lexer::token_type::value_unsigned));
|
|
|
|
CHECK((scan_string("-0") == json::lexer::token_type::value_integer));
|
|
CHECK((scan_string("-1") == json::lexer::token_type::value_integer));
|
|
|
|
CHECK((scan_string("1.1") == json::lexer::token_type::value_float));
|
|
CHECK((scan_string("-1.1") == json::lexer::token_type::value_float));
|
|
CHECK((scan_string("1E10") == json::lexer::token_type::value_float));
|
|
}
|
|
|
|
SECTION("whitespace")
|
|
{
|
|
// result is end_of_input, because not token is following
|
|
CHECK((scan_string(" ") == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string("\t") == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string("\n") == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string("\r") == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string(" \t\n\r\n\t ") == json::lexer::token_type::end_of_input));
|
|
}
|
|
}
|
|
|
|
SECTION("token_type_name")
|
|
{
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::uninitialized)) == "<uninitialized>"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::literal_true)) == "true literal"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::literal_false)) == "false literal"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::literal_null)) == "null literal"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::value_string)) == "string literal"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::value_unsigned)) == "number literal"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::value_integer)) == "number literal"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::value_float)) == "number literal"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::begin_array)) == "'['"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::begin_object)) == "'{'"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::end_array)) == "']'"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::end_object)) == "'}'"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::name_separator)) == "':'"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::value_separator)) == "','"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::parse_error)) == "<parse error>"));
|
|
CHECK((std::string(json::lexer::token_type_name(json::lexer::token_type::end_of_input)) == "end of input"));
|
|
}
|
|
|
|
SECTION("parse errors on first character")
|
|
{
|
|
for (int c = 1; c < 128; ++c)
|
|
{
|
|
// create string from the ASCII code
|
|
const auto s = std::string(1, static_cast<char>(c));
|
|
// store scan() result
|
|
const auto res = scan_string(s.c_str());
|
|
|
|
CAPTURE(s)
|
|
|
|
switch (c)
|
|
{
|
|
// single characters that are valid tokens
|
|
case ('['):
|
|
case (']'):
|
|
case ('{'):
|
|
case ('}'):
|
|
case (','):
|
|
case (':'):
|
|
case ('0'):
|
|
case ('1'):
|
|
case ('2'):
|
|
case ('3'):
|
|
case ('4'):
|
|
case ('5'):
|
|
case ('6'):
|
|
case ('7'):
|
|
case ('8'):
|
|
case ('9'):
|
|
{
|
|
CHECK((res != json::lexer::token_type::parse_error));
|
|
break;
|
|
}
|
|
|
|
// whitespace
|
|
case (' '):
|
|
case ('\t'):
|
|
case ('\n'):
|
|
case ('\r'):
|
|
{
|
|
CHECK((res == json::lexer::token_type::end_of_input));
|
|
break;
|
|
}
|
|
|
|
// anything else is not expected
|
|
default:
|
|
{
|
|
CHECK((res == json::lexer::token_type::parse_error));
|
|
break;
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
SECTION("very large string")
|
|
{
|
|
// strings larger than 1024 bytes yield a resize of the lexer's yytext buffer
|
|
std::string s("\"");
|
|
s += std::string(2048, 'x');
|
|
s += "\"";
|
|
CHECK((scan_string(s.c_str()) == json::lexer::token_type::value_string));
|
|
}
|
|
|
|
SECTION("fail on comments")
|
|
{
|
|
CHECK((scan_string("/", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/", false) == "invalid literal");
|
|
|
|
CHECK((scan_string("/!", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/!", false) == "invalid literal");
|
|
CHECK((scan_string("/*", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/*", false) == "invalid literal");
|
|
CHECK((scan_string("/**", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/**", false) == "invalid literal");
|
|
|
|
CHECK((scan_string("//", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("//", false) == "invalid literal");
|
|
CHECK((scan_string("/**/", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/**/", false) == "invalid literal");
|
|
CHECK((scan_string("/** /", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/** /", false) == "invalid literal");
|
|
|
|
CHECK((scan_string("/***/", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/***/", false) == "invalid literal");
|
|
CHECK((scan_string("/* true */", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/* true */", false) == "invalid literal");
|
|
CHECK((scan_string("/*/**/", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/*/**/", false) == "invalid literal");
|
|
CHECK((scan_string("/*/* */", false) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/*/* */", false) == "invalid literal");
|
|
}
|
|
|
|
SECTION("ignore comments")
|
|
{
|
|
CHECK((scan_string("/", true) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/", true) == "invalid comment; expecting '/' or '*' after '/'");
|
|
|
|
CHECK((scan_string("/!", true) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/!", true) == "invalid comment; expecting '/' or '*' after '/'");
|
|
CHECK((scan_string("/*", true) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/*", true) == "invalid comment; missing closing '*/'");
|
|
CHECK((scan_string("/**", true) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/**", true) == "invalid comment; missing closing '*/'");
|
|
|
|
CHECK((scan_string("//", true) == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string("/**/", true) == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string("/** /", true) == json::lexer::token_type::parse_error));
|
|
CHECK(get_error_message("/** /", true) == "invalid comment; missing closing '*/'");
|
|
|
|
CHECK((scan_string("/***/", true) == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string("/* true */", true) == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string("/*/**/", true) == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string("/*/* */", true) == json::lexer::token_type::end_of_input));
|
|
|
|
CHECK((scan_string("//\n//\n", true) == json::lexer::token_type::end_of_input));
|
|
CHECK((scan_string("/**//**//**/", true) == json::lexer::token_type::end_of_input));
|
|
}
|
|
}
|
|
|
|
TEST_CASE("lexer number fast path")
|
|
{
|
|
// The contiguous fast path (used for pointer/string input) must agree with
|
|
// the streaming byte path (used for std::istream) on token type, numeric
|
|
// value, and round-trip text for every well-formed number, and reject the
|
|
// same malformed numbers with the same message.
|
|
SECTION("contiguous vs streaming parity")
|
|
{
|
|
const std::vector<std::string> numbers =
|
|
{
|
|
"0", "-0", "1", "-1", "42", "-42", "10", "100", "1234567890",
|
|
"0.0", "-0.0", "3.14", "-3.14", "0.5", "-0.001", "123.456789",
|
|
"1e0", "1E0", "1e10", "1e-10", "1e+10", "1.5e3", "-2.5E-4",
|
|
"9223372036854775807", // INT64_MAX -> unsigned
|
|
"9223372036854775808", // INT64_MAX + 1 -> unsigned
|
|
"18446744073709551615", // UINT64_MAX -> unsigned
|
|
"18446744073709551616", // UINT64_MAX + 1 -> float
|
|
"-9223372036854775808", // INT64_MIN -> integer
|
|
"-9223372036854775809", // INT64_MIN - 1 -> float
|
|
"123456789012345678901234567890", // huge -> float
|
|
"0.30000000000000004", "2.2250738585072014e-308", "1e308",
|
|
// high-precision / wide-exponent values that exercise the
|
|
// Eisel-Lemire path beyond the Clinger subset
|
|
"1.7976931348623157e308", "1.2345678901234567e-250",
|
|
"9007199254740993", "5e-324", "1e-320"
|
|
};
|
|
|
|
for (const auto& n : numbers)
|
|
{
|
|
const std::string doc = "[" + n + "]";
|
|
|
|
// contiguous fast path
|
|
const json a = json::parse(doc);
|
|
// streaming byte path
|
|
std::stringstream ss(doc);
|
|
const json b = json::parse(ss);
|
|
|
|
CAPTURE(n);
|
|
CHECK(a == b);
|
|
CHECK(a.dump() == b.dump());
|
|
CHECK(a[0].type() == b[0].type());
|
|
}
|
|
}
|
|
|
|
SECTION("significant digits around Clinger's fast path")
|
|
{
|
|
// Clinger's fast path needs a significand of at most 2^53, which
|
|
// tokens with 17 or more significant digits exceed. The conversion
|
|
// splits the token at the positions the scanners recorded, so leading
|
|
// zeros must not count as digits - "0.1234567890123456" has 16
|
|
// significant digits, not 17 - and both scanners must agree.
|
|
const std::vector<std::string> numbers =
|
|
{
|
|
"1234567890123456", // 16 significant digits
|
|
"12345678901234567", // 17
|
|
"123456789012345678", // 18
|
|
"0.1234567890123456", // 16: the leading "0" is not significant
|
|
"0.12345678901234567", // 17
|
|
"0.00000000000000001", // 1, in a long token
|
|
"0.000000000000000012345678901234", // 14, in a long token
|
|
"-0.0000000000000000000001", // 1, negative
|
|
"1.0000000000000000", // 17: trailing zeros are significant here
|
|
"10000000000000000", // 17
|
|
"9007199254740992", // 2^53
|
|
"9007199254740993", // 2^53 + 1
|
|
"-65.613616999999977", // canada.json shape
|
|
"1.2345678901234567e-250", // 17 with an exponent
|
|
"1.234567890123456e-250", // 16 with an exponent
|
|
"1e10", "0.0", "-0.0", "0e0", "0.000123"
|
|
};
|
|
|
|
for (const auto& n : numbers)
|
|
{
|
|
CAPTURE(n);
|
|
const std::string doc = "[" + n + "]";
|
|
|
|
const json a = json::parse(doc); // contiguous fast path
|
|
std::stringstream ss(doc);
|
|
const json b = json::parse(ss); // streaming byte path
|
|
|
|
CHECK(a[0].type() == b[0].type());
|
|
CHECK(a == b);
|
|
|
|
if (a[0].is_number_float())
|
|
{
|
|
const double expected = std::strtod(n.c_str(), nullptr);
|
|
CHECK(a[0].get<double>() == expected);
|
|
CHECK(b[0].get<double>() == expected);
|
|
}
|
|
}
|
|
}
|
|
|
|
SECTION("token type classification")
|
|
{
|
|
CHECK((scan_string("0") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("-1") == json::lexer::token_type::value_integer));
|
|
CHECK((scan_string("1.5") == json::lexer::token_type::value_float));
|
|
CHECK((scan_string("1e5") == json::lexer::token_type::value_float));
|
|
CHECK((scan_string("18446744073709551615") == json::lexer::token_type::value_unsigned));
|
|
CHECK((scan_string("18446744073709551616") == json::lexer::token_type::value_float));
|
|
CHECK((scan_string("-9223372036854775808") == json::lexer::token_type::value_integer));
|
|
CHECK((scan_string("-9223372036854775809") == json::lexer::token_type::value_float));
|
|
}
|
|
|
|
SECTION("malformed numbers are rejected identically")
|
|
{
|
|
for (const char* bad :
|
|
{"-", "1.", "1e", "1e+", "1.2e", "01", "-01", "1..2", "1.2.3"
|
|
})
|
|
{
|
|
CAPTURE(bad);
|
|
// the contiguous fast path must decline and let the byte path report
|
|
const std::string doc = std::string("[") + bad + "]";
|
|
CHECK_FALSE(json::accept(doc));
|
|
std::stringstream ss(doc);
|
|
CHECK_FALSE(json::accept(ss));
|
|
}
|
|
}
|
|
|
|
#if !defined(JSON_NOEXCEPTION)
|
|
// these sections parse invalid input, which aborts when exceptions are off
|
|
SECTION("exhaustive grammar parity with the streaming path")
|
|
{
|
|
// The JSON number grammar is encoded twice: once as the scan_number()
|
|
// state machine and once as the contiguous fast path. Enumerate every
|
|
// short string over the number alphabet and require the two encodings to
|
|
// agree exactly - on acceptance, on the reported error, and on the parsed
|
|
// value - so they cannot drift apart.
|
|
const std::string alphabet = "01.eE+-";
|
|
|
|
// full outcome of parsing @a doc, so a mismatch in type, value, or error
|
|
// message is caught, not just a mismatch in acceptance
|
|
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
|
|
{
|
|
try
|
|
{
|
|
if (streaming)
|
|
{
|
|
std::stringstream ss(doc);
|
|
const json j = json::parse(ss);
|
|
return std::string(j[0].type_name()) + '|' + j.dump();
|
|
}
|
|
const json j = json::parse(doc);
|
|
return std::string(j[0].type_name()) + '|' + j.dump();
|
|
}
|
|
catch (const json::parse_error& e)
|
|
{
|
|
return {e.what()};
|
|
}
|
|
};
|
|
|
|
std::vector<std::string> mismatches;
|
|
std::vector<std::string> tokens{""};
|
|
for (std::size_t length = 1; length <= 4; ++length)
|
|
{
|
|
std::vector<std::string> next;
|
|
next.reserve(tokens.size() * alphabet.size());
|
|
for (const auto& prefix : tokens)
|
|
{
|
|
for (const char c : alphabet)
|
|
{
|
|
next.push_back(prefix + c);
|
|
}
|
|
}
|
|
tokens = next;
|
|
|
|
for (const auto& token : tokens)
|
|
{
|
|
const std::string doc = "[" + token + "]";
|
|
if (outcome(doc, false) != outcome(doc, true))
|
|
{
|
|
mismatches.push_back(doc);
|
|
}
|
|
}
|
|
}
|
|
|
|
// 7 + 49 + 343 + 2401 tokens
|
|
CHECK(tokens.size() == 2401);
|
|
CAPTURE(mismatches);
|
|
CHECK(mismatches.empty());
|
|
}
|
|
|
|
SECTION("error positions match the streaming path")
|
|
{
|
|
// Rejecting identically is not enough: the fast path must also report the
|
|
// error at the same position as the byte path. A number directly followed
|
|
// by a newline is the interesting case, because the byte path reaches the
|
|
// newline (which resets the column) and then ungets it.
|
|
// returns the parse_error message, or "" if the document parsed
|
|
const auto contiguous_error = [](const std::string & doc) -> std::string
|
|
{
|
|
try
|
|
{
|
|
const json j = json::parse(doc);
|
|
static_cast<void>(j);
|
|
}
|
|
catch (const json::parse_error& e)
|
|
{
|
|
return {e.what()};
|
|
}
|
|
return {};
|
|
};
|
|
const auto streaming_error = [](const std::string & doc) -> std::string
|
|
{
|
|
try
|
|
{
|
|
std::stringstream ss(doc);
|
|
const json j = json::parse(ss);
|
|
static_cast<void>(j);
|
|
}
|
|
catch (const json::parse_error& e)
|
|
{
|
|
return {e.what()};
|
|
}
|
|
return {};
|
|
};
|
|
|
|
for (const char* bad :
|
|
{"[01\n]", "[00\n]", "[-01\n]", "{1\n}", "[1\n2]", "[1.2.3\n]",
|
|
"[1 \n2]", "[\n1\n2]", "1\n2", "[01\r\n]", "[1e\n]", "[-\n]"
|
|
})
|
|
{
|
|
CAPTURE(bad);
|
|
const std::string doc = bad;
|
|
const std::string contiguous_what = contiguous_error(doc);
|
|
|
|
CHECK_FALSE(contiguous_what.empty());
|
|
CHECK(contiguous_what == streaming_error(doc));
|
|
}
|
|
|
|
// A number terminated by a newline must report the same position as the
|
|
// same number terminated by anything else: scan_number() reads the
|
|
// terminator and ungets it, so the reported column is the one reached
|
|
// after the number's last character - not the 0 that an unget() across
|
|
// the newline used to leave behind.
|
|
CHECK(contiguous_error("[01\n]") == contiguous_error("[01 ]"));
|
|
CHECK(contiguous_error("[01\n]") ==
|
|
"[json.exception.parse_error.101] parse error at line 1, column 3: "
|
|
"syntax error while parsing array - unexpected number literal; expected ']'");
|
|
|
|
// the same for a multi-character token, where the column of the last
|
|
// character (the '3' of "-2.5e3") differs from the column it starts at
|
|
CHECK(contiguous_error("null -2.5e3\nfalse") == contiguous_error("null -2.5e3 false"));
|
|
CHECK(contiguous_error("null -2.5e3\nfalse") ==
|
|
"[json.exception.parse_error.101] parse error at line 1, column 11: "
|
|
"syntax error while parsing value - unexpected number literal; expected end of input");
|
|
}
|
|
#endif
|
|
}
|
|
|
|
TEST_CASE("lexer string fast path")
|
|
{
|
|
// Build a byte string from explicit values: a hex escape in a string
|
|
// literal swallows every following hex digit, which makes sequences like
|
|
// "\xC3\xA9b" mean something other than they look like.
|
|
const auto bytes = [](std::initializer_list<int> values)
|
|
{
|
|
std::string result;
|
|
for (const int value : values)
|
|
{
|
|
result.push_back(static_cast<char>(value));
|
|
}
|
|
return result;
|
|
};
|
|
|
|
#if !defined(JSON_NOEXCEPTION)
|
|
// the full outcome of parsing @a doc: the parsed value, or the exact error
|
|
// message, so a mismatch in either is caught. Only usable with exceptions
|
|
// on: parsing invalid input aborts when they are off.
|
|
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
|
|
{
|
|
try
|
|
{
|
|
if (streaming)
|
|
{
|
|
std::stringstream ss(doc);
|
|
const json j = json::parse(ss);
|
|
return j.dump();
|
|
}
|
|
const json j = json::parse(doc);
|
|
return j.dump();
|
|
}
|
|
// not just parse_error: if a bulk scanner ever let ill-formed UTF-8
|
|
// through, dump() would throw type_error.316, and that has to surface
|
|
// as a reported mismatch rather than as an uncaught exception
|
|
catch (const json::exception& e)
|
|
{
|
|
return {e.what()};
|
|
}
|
|
};
|
|
#endif
|
|
|
|
// once at the start of the string, once past the first 8-byte SWAR word, so
|
|
// the bulk scanner sees each case with and without a run behind it
|
|
const std::vector<std::size_t> offsets{0, 9};
|
|
|
|
#if !defined(JSON_NOEXCEPTION)
|
|
SECTION("exhaustive contiguous vs streaming parity")
|
|
{
|
|
// ordinary ASCII, both specials, a control byte, characters that make
|
|
// the preceding backslash a valid escape, a UTF-8 lead byte of each
|
|
// length, a continuation byte, and a byte that is never valid
|
|
const std::vector<std::string> alphabet =
|
|
{
|
|
"a", "\"", "\\", "n", "u", "0", bytes({0x01}),
|
|
bytes({0xC3}), bytes({0xA9}), bytes({0xE4}), bytes({0xF0}),
|
|
bytes({0x80}), bytes({0xFF})
|
|
};
|
|
|
|
std::vector<std::string> mismatches;
|
|
std::vector<std::string> tokens{""};
|
|
for (std::size_t length = 1; length <= 3; ++length)
|
|
{
|
|
std::vector<std::string> next;
|
|
next.reserve(tokens.size() * alphabet.size());
|
|
for (const auto& prefix : tokens)
|
|
{
|
|
for (const auto& symbol : alphabet)
|
|
{
|
|
next.push_back(prefix + symbol);
|
|
}
|
|
}
|
|
tokens = next;
|
|
|
|
for (const auto& token : tokens)
|
|
{
|
|
for (const std::size_t offset : offsets)
|
|
{
|
|
const std::string doc = "[\"" + std::string(offset, 'a') + token + "\"]";
|
|
if (outcome(doc, false) != outcome(doc, true))
|
|
{
|
|
mismatches.push_back(doc);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
// 13 + 169 + 2197 tokens, each at two offsets
|
|
CHECK(tokens.size() == 2197);
|
|
CAPTURE(mismatches);
|
|
CHECK(mismatches.empty());
|
|
}
|
|
|
|
SECTION("special bytes at every offset of the SWAR stride")
|
|
{
|
|
// The bulk scanner consumes 8 bytes at a time and then a tail; place
|
|
// every kind of byte that ends a run at each offset across two words,
|
|
// so multibyte sequences also straddle the word boundary.
|
|
const std::vector<std::string> specials =
|
|
{
|
|
"\"", "\\", bytes({0x01}), bytes({0x1F}), bytes({0x7F}),
|
|
bytes({0xC3, 0xA9}), bytes({0xE4, 0xB8, 0xAD}), bytes({0xF0, 0x9F, 0x98, 0x80}),
|
|
bytes({0xFF}), bytes({0xC3}), bytes({0xE4, 0xB8})
|
|
};
|
|
|
|
std::vector<std::string> mismatches;
|
|
for (std::size_t offset = 0; offset <= 17; ++offset)
|
|
{
|
|
for (const auto& special : specials)
|
|
{
|
|
const std::string doc = "[\"" + std::string(offset, 'a') + special + "\"]";
|
|
if (outcome(doc, false) != outcome(doc, true))
|
|
{
|
|
mismatches.push_back(doc);
|
|
}
|
|
}
|
|
}
|
|
CAPTURE(mismatches);
|
|
CHECK(mismatches.empty());
|
|
}
|
|
#endif
|
|
|
|
// json::accept() never throws, so the ranges stay covered without exceptions
|
|
SECTION("UTF-8 ranges are accepted and rejected as documented")
|
|
{
|
|
// The bulk validator must accept exactly what the byte-at-a-time
|
|
// scanner accepts, so pin the boundaries of every range it recognizes.
|
|
// aggregate, only ever brace-initialized below; default member
|
|
// initializers would stop it being an aggregate in C++11
|
|
struct utf8_case // NOLINT(cppcoreguidelines-pro-type-member-init,hicpp-member-init)
|
|
{
|
|
std::string sequence;
|
|
bool valid;
|
|
const char* description;
|
|
};
|
|
const std::vector<utf8_case> cases =
|
|
{
|
|
{bytes({0xC2, 0x80}), true, "U+0080, shortest two-byte"},
|
|
{bytes({0xDF, 0xBF}), true, "U+07FF, longest two-byte"},
|
|
{bytes({0xC1, 0xBF}), false, "overlong two-byte"},
|
|
{bytes({0xC2, 0x7F}), false, "two-byte with bad continuation"},
|
|
{bytes({0xE0, 0xA0, 0x80}), true, "U+0800, shortest three-byte"},
|
|
{bytes({0xE0, 0x9F, 0xBF}), false, "overlong three-byte"},
|
|
{bytes({0xED, 0x9F, 0xBF}), true, "U+D7FF, just below the surrogates"},
|
|
{bytes({0xED, 0xA0, 0x80}), false, "surrogate U+D800"},
|
|
{bytes({0xED, 0xBF, 0xBF}), false, "surrogate U+DFFF"},
|
|
{bytes({0xEE, 0x80, 0x80}), true, "U+E000, just above the surrogates"},
|
|
{bytes({0xEF, 0xBF, 0xBF}), true, "U+FFFF"},
|
|
{bytes({0xF0, 0x90, 0x80, 0x80}), true, "U+10000, shortest four-byte"},
|
|
{bytes({0xF0, 0x8F, 0xBF, 0xBF}), false, "overlong four-byte"},
|
|
{bytes({0xF4, 0x8F, 0xBF, 0xBF}), true, "U+10FFFF, highest code point"},
|
|
{bytes({0xF4, 0x90, 0x80, 0x80}), false, "above U+10FFFF"},
|
|
{bytes({0xF5, 0x80, 0x80, 0x80}), false, "lead byte out of range"},
|
|
{bytes({0x80}), false, "bare continuation byte"},
|
|
{bytes({0xFF}), false, "byte that never appears in UTF-8"},
|
|
{bytes({0xC3}), false, "truncated two-byte"},
|
|
{bytes({0xE4, 0xB8}), false, "truncated three-byte"},
|
|
{bytes({0xF0, 0x9F, 0x98}), false, "truncated four-byte"}
|
|
};
|
|
|
|
for (const auto& test_case : cases)
|
|
{
|
|
CAPTURE(test_case.description);
|
|
for (const std::size_t offset : offsets)
|
|
{
|
|
CAPTURE(offset);
|
|
const std::string doc = "[\"" + std::string(offset, 'a') + test_case.sequence + "\"]";
|
|
CHECK(json::accept(doc) == test_case.valid);
|
|
#if !defined(JSON_NOEXCEPTION)
|
|
CHECK(outcome(doc, false) == outcome(doc, true));
|
|
#endif
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
TEST_CASE("lexer escape fast path")
|
|
{
|
|
// json::accept() never throws, so this section stays covered without
|
|
// exceptions; it pins which of the cases below are valid/invalid and
|
|
// checks the contiguous and streaming paths agree on that classification.
|
|
SECTION("accept() parity")
|
|
{
|
|
const std::vector<std::pair<std::string, bool>> cases =
|
|
{
|
|
{"\\u0041", true}, {"\\u00e4", true}, {"\\u00E4", true},
|
|
{"\\uD83D\\uDE00", true},
|
|
{"\\u12", false}, {"\\u12G4", false}, {"\\uXYZW", false},
|
|
{"\\uD800", false}, {"\\uD800A", false}, {"\\uD800\\u0041", false},
|
|
{"\\uDC00", false}, {"\\u", false}
|
|
};
|
|
|
|
for (const auto& c : cases)
|
|
{
|
|
for (const std::size_t offset :
|
|
{
|
|
std::size_t{0}, std::size_t{9}
|
|
})
|
|
{
|
|
const std::string doc = "[\"" + std::string(offset, 'a') + c.first + "\"]";
|
|
CAPTURE(doc);
|
|
CHECK(json::accept(doc) == c.second);
|
|
std::stringstream ss(doc);
|
|
CHECK(json::accept(ss) == c.second);
|
|
}
|
|
}
|
|
}
|
|
|
|
#if !defined(JSON_NOEXCEPTION)
|
|
// the full outcome of parsing @a doc: the parsed value, or the exact
|
|
// error message, so a mismatch in either is caught
|
|
const auto outcome = [](const std::string & doc, bool streaming) -> std::string
|
|
{
|
|
try
|
|
{
|
|
if (streaming)
|
|
{
|
|
std::stringstream ss(doc);
|
|
const json j = json::parse(ss);
|
|
return j.dump();
|
|
}
|
|
const json j = json::parse(doc);
|
|
return j.dump();
|
|
}
|
|
catch (const json::exception& e)
|
|
{
|
|
return {e.what()};
|
|
}
|
|
};
|
|
|
|
SECTION("contiguous vs streaming parity")
|
|
{
|
|
const std::vector<std::string> escapes =
|
|
{
|
|
"\\u0041", // "A"
|
|
"\\u00e4", // "ä" (lowercase hex)
|
|
"\\u00E4", // "ä" (uppercase hex)
|
|
"\\uD83D\\uDE00", // valid surrogate pair (an emoji)
|
|
"\\u12", // truncated: only 2 hex digits before the closing quote
|
|
"\\u12G4", // invalid hex digit at the 3rd position
|
|
"\\uXYZW", // all 4 bytes invalid
|
|
"\\uD800", // lone high surrogate, string ends right after
|
|
"\\uD800A", // high surrogate not followed by another \u escape
|
|
"\\uD800\\u0041", // high surrogate followed by \u, but not a low surrogate
|
|
"\\uDC00", // lone low surrogate
|
|
"\\u", // '\u' with nothing after (closing quote right away)
|
|
};
|
|
|
|
// once at the start of the string and once past the first 8-byte SWAR
|
|
// word of the outer string_bulk_run, so the escape is reached both
|
|
// right after the opening quote and mid-run
|
|
for (const auto& escape : escapes)
|
|
{
|
|
for (const std::size_t offset :
|
|
{
|
|
std::size_t{0}, std::size_t{9}
|
|
})
|
|
{
|
|
const std::string doc = "[\"" + std::string(offset, 'a') + escape + "\"]";
|
|
CAPTURE(doc);
|
|
CHECK(outcome(doc, false) == outcome(doc, true));
|
|
}
|
|
|
|
// the escape is the last thing before end of input: no closing
|
|
// quote at all
|
|
const std::string truncated_doc = "[\"" + escape;
|
|
CAPTURE(truncated_doc);
|
|
CHECK(outcome(truncated_doc, false) == outcome(truncated_doc, true));
|
|
}
|
|
}
|
|
|
|
SECTION("truncated \\u escape at every distance from the end of input")
|
|
{
|
|
// ia.bulk_remaining() must correctly report fewer than 4 bytes for
|
|
// every possible count of trailing hex-looking bytes (0, 1, 2, or 3)
|
|
// before end of input, so the fast path declines and the byte path
|
|
// alone reports the "must be followed by 4 hex digits" error, at the
|
|
// same position, in every case
|
|
for (const std::string& tail :
|
|
{
|
|
std::string{}, std::string("1"), std::string("12"), std::string("123")
|
|
})
|
|
{
|
|
const std::string doc = "[\"\\u" + tail;
|
|
CAPTURE(doc);
|
|
CHECK(outcome(doc, false) == outcome(doc, true));
|
|
CHECK(outcome(doc, false).find("must be followed by 4 hex digits") != std::string::npos);
|
|
}
|
|
}
|
|
|
|
SECTION("invalid hex digit at every position of the 4")
|
|
{
|
|
// the fast path must decline for *any* invalid byte among the 4, not
|
|
// just the first, and the byte path must then stop at exactly that
|
|
// position - same as it always has
|
|
for (std::size_t bad_pos = 0; bad_pos < 4; ++bad_pos)
|
|
{
|
|
std::string digits = "1234";
|
|
digits[bad_pos] = 'g'; // not a hex digit
|
|
const std::string doc = "[\"\\u" + digits + "\"]";
|
|
CAPTURE(doc);
|
|
CHECK(outcome(doc, false) == outcome(doc, true));
|
|
CHECK(outcome(doc, false).find("must be followed by 4 hex digits") != std::string::npos);
|
|
}
|
|
}
|
|
|
|
SECTION("random escapes")
|
|
{
|
|
// A seeded PRNG builds the 4 bytes following `\u` from a mix of hex
|
|
// digits and non-hex bytes, at varying distances from the start of
|
|
// the string, to compare the two scanners on many more shapes than
|
|
// are practical to enumerate by hand.
|
|
std::mt19937 gen(7654321); // NOLINT(cert-msc32-c,cert-msc51-cpp)
|
|
const std::string hex_alphabet = "0123456789AaBbCcDdEeFf";
|
|
std::uniform_int_distribution<std::size_t> pick_hex(0, hex_alphabet.size() - 1);
|
|
std::uniform_int_distribution<int> pick_byte(1, 255); // never NUL
|
|
std::uniform_int_distribution<int> pick_is_hex(0, 4); // 4-in-5 chance of a hex digit
|
|
std::uniform_int_distribution<std::size_t> pick_offset(0, 12);
|
|
|
|
std::vector<std::string> mismatches;
|
|
for (int iter = 0; iter < 3000; ++iter)
|
|
{
|
|
std::string digits;
|
|
for (int i = 0; i < 4; ++i)
|
|
{
|
|
if (pick_is_hex(gen) != 0)
|
|
{
|
|
digits += hex_alphabet[pick_hex(gen)];
|
|
}
|
|
else
|
|
{
|
|
char c = static_cast<char>(pick_byte(gen));
|
|
if (c == '"' || c == '\\')
|
|
{
|
|
// keep the string well-formed apart from the escape
|
|
// itself, so any mismatch is attributable to the \u
|
|
// handling and not to an unrelated quote/escape
|
|
c = 'z';
|
|
}
|
|
digits += c;
|
|
}
|
|
}
|
|
const std::string doc = "[\"" + std::string(pick_offset(gen), 'a') + "\\u" + digits + "\"]";
|
|
if (outcome(doc, false) != outcome(doc, true))
|
|
{
|
|
mismatches.push_back(doc);
|
|
}
|
|
}
|
|
CAPTURE(mismatches);
|
|
CHECK(mismatches.empty());
|
|
}
|
|
#endif
|
|
}
|
|
|
|
namespace
|
|
{
|
|
// the index of the decimal point (or npos) and of the end of the mantissa of a
|
|
// number token, which the lexer records while scanning it
|
|
std::pair<std::size_t, std::size_t> float_token_layout(const std::string& s)
|
|
{
|
|
std::size_t dot = std::string::npos;
|
|
std::size_t mantissa_end = s.size();
|
|
for (std::size_t i = 0; i < s.size(); ++i)
|
|
{
|
|
if (s[i] == '.')
|
|
{
|
|
dot = i;
|
|
}
|
|
else if (s[i] == 'e' || s[i] == 'E')
|
|
{
|
|
mantissa_end = i;
|
|
break;
|
|
}
|
|
}
|
|
return {dot, mantissa_end};
|
|
}
|
|
|
|
template<typename FloatType>
|
|
FloatType parse_native(const std::string& s)
|
|
{
|
|
const auto layout = float_token_layout(s);
|
|
return nlohmann::detail::parse_float_native<FloatType>(s.data(), s.data() + s.size(), layout.first, layout.second);
|
|
}
|
|
|
|
std::uint64_t bits_of(double d)
|
|
{
|
|
std::uint64_t b = 0;
|
|
std::memcpy(&b, &d, sizeof(b));
|
|
return b;
|
|
}
|
|
|
|
std::uint32_t bits_of(float f)
|
|
{
|
|
std::uint32_t b = 0;
|
|
std::memcpy(&b, &f, sizeof(b));
|
|
return b;
|
|
}
|
|
|
|
std::uint64_t native_bits64(const std::string& s)
|
|
{
|
|
return bits_of(parse_native<double>(s));
|
|
}
|
|
|
|
std::uint32_t native_bits32(const std::string& s)
|
|
{
|
|
return bits_of(parse_native<float>(s));
|
|
}
|
|
} // namespace
|
|
|
|
TEST_CASE("parse_float_native rounds correctly")
|
|
{
|
|
SECTION("double")
|
|
{
|
|
CHECK(native_bits64("1.5") == 0x3FF8000000000000u);
|
|
CHECK(native_bits64("0.1") == 0x3FB999999999999Au);
|
|
CHECK(native_bits64("-0.0") == 0x8000000000000000u);
|
|
CHECK(native_bits64("0e999999999999999999999") == 0u);
|
|
// 2^53 + 1 is exactly between two doubles: ties to even, unless more digits follow
|
|
CHECK(native_bits64("9007199254740993") == 0x4340000000000000u);
|
|
CHECK(native_bits64("9007199254740993.0000000000000000001") == 0x4340000000000001u);
|
|
CHECK(native_bits64("9007199254740992.9999999999999999999") == 0x4340000000000000u);
|
|
// 1 + 2^-53 exactly (a tie), and one unit in the 55th digit around it
|
|
CHECK(native_bits64("1.00000000000000011102230246251565404236316680908203125") == 0x3FF0000000000000u);
|
|
CHECK(native_bits64("1.00000000000000011102230246251565404236316680908203126") == 0x3FF0000000000001u);
|
|
CHECK(native_bits64("1.00000000000000011102230246251565404236316680908203124") == 0x3FF0000000000000u);
|
|
// subnormal and overflow boundaries
|
|
CHECK(native_bits64("2.4703282292062327e-324") == 0u);
|
|
CHECK(native_bits64("2.4703282292062328e-324") == 1u);
|
|
CHECK(native_bits64("2.2250738585072011e-308") == 0x000FFFFFFFFFFFFFu);
|
|
CHECK(native_bits64("2.2250738585072012e-308") == 0x0010000000000000u);
|
|
CHECK(native_bits64("1.7976931348623157e308") == 0x7FEFFFFFFFFFFFFFu);
|
|
CHECK(native_bits64("1.7976931348623159e308") == 0x7FF0000000000000u);
|
|
CHECK(native_bits64("-1e400") == 0xFFF0000000000000u);
|
|
CHECK(native_bits64("-1e-400") == 0x8000000000000000u);
|
|
// exponents and zeros far beyond the range cancel out
|
|
CHECK(native_bits64("0." + std::string(1000, '0') + "1e1001") == 0x3FF0000000000000u);
|
|
CHECK(native_bits64("1" + std::string(1000, '0') + "e-1000") == 0x3FF0000000000000u);
|
|
CHECK(native_bits64("1e-99999999999999999999999") == 0u);
|
|
CHECK(native_bits64("1E+99999999999999999999999") == 0x7FF0000000000000u);
|
|
// more digits than any midpoint has (769): only whether a nonzero digit follows matters
|
|
const std::string tie = "1.00000000000000011102230246251565404236316680908203125";
|
|
CHECK(native_bits64(tie + std::string(800, '0')) == 0x3FF0000000000000u);
|
|
CHECK(native_bits64(tie + std::string(800, '0') + "1") == 0x3FF0000000000001u);
|
|
}
|
|
|
|
SECTION("float")
|
|
{
|
|
CHECK(native_bits32("1.5") == 0x3FC00000u);
|
|
CHECK(native_bits32("0.1") == 0x3DCCCCCDu);
|
|
CHECK(native_bits32("-0.0") == 0x80000000u);
|
|
// 2^24 + 1 is exactly between two floats
|
|
CHECK(native_bits32("16777217") == 0x4B800000u);
|
|
CHECK(native_bits32("16777217.000000000000000000001") == 0x4B800001u);
|
|
CHECK(native_bits32("16777218.999999999999999999999") == 0x4B800001u);
|
|
CHECK(native_bits32("16777219") == 0x4B800002u);
|
|
// subnormal and overflow boundaries
|
|
CHECK(native_bits32("3.4028235677973366e38") == 0x7F7FFFFFu);
|
|
CHECK(native_bits32("3.4028235677973367e38") == 0x7F800000u);
|
|
CHECK(native_bits32("7.006492321624085e-46") == 0u);
|
|
CHECK(native_bits32("7.006492321624086e-46") == 1u);
|
|
CHECK(native_bits32("1.1754942e-38") == 0x007FFFFFu);
|
|
CHECK(native_bits32("-1.17549435e-38") == 0x80800000u);
|
|
CHECK(native_bits32("1e39") == 0x7F800000u);
|
|
CHECK(native_bits32("-1e-50") == 0x80000000u);
|
|
// not rounded through double: its double would round to another float
|
|
CHECK(native_bits32("1.00000005960464477539062500000000001") == 0x3F800001u);
|
|
CHECK(native_bits32("9007199254740993") == 0x5A000000u);
|
|
}
|
|
|
|
SECTION("the conversion shared with other parsers")
|
|
{
|
|
// convert_float() gives the lexer's results, for every type
|
|
const std::vector<std::string> tokens =
|
|
{
|
|
"0", "-0.0", "1.5", "0.1", "1e-400", "-2.5E+3", "123456789012345678901234567890",
|
|
"9007199254740993.0000000000000000001", "4.9406564584124654e-324"
|
|
};
|
|
using float_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
|
|
using long_double_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, long double>;
|
|
for (const auto& t : tokens)
|
|
{
|
|
CAPTURE(t);
|
|
const auto layout = float_token_layout(t);
|
|
const char* const first = t.data();
|
|
const char* const last = first + t.size();
|
|
const auto d = nlohmann::detail::convert_float<double>(first, last, layout.first, layout.second);
|
|
const auto f = nlohmann::detail::convert_float<float>(first, last, layout.first, layout.second);
|
|
const auto ld = nlohmann::detail::convert_float<long double>(first, last, layout.first, layout.second);
|
|
CHECK(bits_of(d) == bits_of(json::parse(t).get<double>()));
|
|
CHECK(bits_of(f) == bits_of(float_json::parse(t).get<float>()));
|
|
CHECK(ld == long_double_json::parse(t).get<long double>());
|
|
}
|
|
}
|
|
}
|
|
|
|
namespace
|
|
{
|
|
// arbitrary-precision unsigned integers, just enough to recompute the table of
|
|
// powers of five (little-endian 32-bit limbs)
|
|
using big_uint = std::vector<std::uint32_t>;
|
|
|
|
void big_trim(big_uint& a)
|
|
{
|
|
while (!a.empty() && a.back() == 0)
|
|
{
|
|
a.pop_back();
|
|
}
|
|
}
|
|
|
|
big_uint big_from(std::uint64_t high, std::uint64_t low)
|
|
{
|
|
big_uint a = {static_cast<std::uint32_t>(low), static_cast<std::uint32_t>(low >> 32u),
|
|
static_cast<std::uint32_t>(high), static_cast<std::uint32_t>(high >> 32u)
|
|
};
|
|
big_trim(a);
|
|
return a;
|
|
}
|
|
|
|
big_uint big_mul(const big_uint& a, const big_uint& b)
|
|
{
|
|
big_uint r(a.size() + b.size(), 0);
|
|
for (std::size_t i = 0; i < a.size(); ++i)
|
|
{
|
|
std::uint64_t carry = 0;
|
|
for (std::size_t j = 0; j < b.size(); ++j)
|
|
{
|
|
const std::uint64_t t = (static_cast<std::uint64_t>(a[i]) * b[j]) + r[i + j] + carry;
|
|
r[i + j] = static_cast<std::uint32_t>(t);
|
|
carry = t >> 32u;
|
|
}
|
|
r[i + b.size()] = static_cast<std::uint32_t>(carry);
|
|
}
|
|
big_trim(r);
|
|
return r;
|
|
}
|
|
|
|
big_uint big_shl(const big_uint& a, std::size_t s)
|
|
{
|
|
big_uint r(s / 32, 0);
|
|
std::uint32_t carry = 0;
|
|
for (const std::uint32_t x : a)
|
|
{
|
|
const std::uint64_t t = static_cast<std::uint64_t>(x) << (s % 32);
|
|
r.push_back(static_cast<std::uint32_t>(t) | carry);
|
|
carry = static_cast<std::uint32_t>(t >> 32u);
|
|
}
|
|
r.push_back(carry);
|
|
big_trim(r);
|
|
return r;
|
|
}
|
|
|
|
// a + 1 (add) or a - 1 (!add, a > 0)
|
|
big_uint big_step(big_uint a, bool add)
|
|
{
|
|
for (auto& x : a)
|
|
{
|
|
const std::uint32_t old = x;
|
|
x = add ? x + 1 : x - 1;
|
|
if ((add && x > old) || (!add && x < old))
|
|
{
|
|
break;
|
|
}
|
|
}
|
|
if (add && (a.empty() || a.back() == 0))
|
|
{
|
|
a.push_back(1);
|
|
}
|
|
big_trim(a);
|
|
return a;
|
|
}
|
|
|
|
bool big_less_equal(const big_uint& a, const big_uint& b)
|
|
{
|
|
if (a.size() != b.size())
|
|
{
|
|
return a.size() < b.size();
|
|
}
|
|
for (std::size_t i = a.size(); i-- > 0;)
|
|
{
|
|
if (a[i] != b[i])
|
|
{
|
|
return a[i] < b[i];
|
|
}
|
|
}
|
|
return true;
|
|
}
|
|
|
|
std::size_t big_bit_length(const big_uint& a)
|
|
{
|
|
std::size_t n = a.size() * 32;
|
|
for (std::uint32_t top = a.back(); (top & 0x80000000u) == 0; top <<= 1u)
|
|
{
|
|
--n;
|
|
}
|
|
return n;
|
|
}
|
|
} // namespace
|
|
|
|
TEST_CASE("Eisel-Lemire float conversion")
|
|
{
|
|
SECTION("the table of powers of five")
|
|
{
|
|
// Recompute every entry the way fast_float's table_generation.py
|
|
// defines it, using only multiplications and comparisons: for q >= 0
|
|
// the most significant 128 bits of 5^q; for q < 0 floor(2^b / 5^-q) + 1
|
|
// for b = z + 127 (q >= -27), or that value for b = 2z + 128 cut to
|
|
// its most significant 128 bits (q < -27), where z is the bit length
|
|
// of 5^-q.
|
|
const auto& table = nlohmann::detail::pow5_128();
|
|
big_uint power5 = {1};
|
|
for (std::int64_t q = 0; q <= nlohmann::detail::pow5_128_largest_power; ++q)
|
|
{
|
|
const auto index = static_cast<std::size_t>(2 * (q - nlohmann::detail::pow5_128_smallest_power));
|
|
const big_uint entry = big_from(table[index], table[index + 1]);
|
|
const std::size_t bits = big_bit_length(power5);
|
|
if (bits <= 128)
|
|
{
|
|
CHECK(entry == big_shl(power5, 128 - bits));
|
|
}
|
|
else
|
|
{
|
|
// floor(5^q / 2^(bits - 128))
|
|
CHECK(big_less_equal(big_shl(entry, bits - 128), power5));
|
|
CHECK_FALSE(big_less_equal(big_shl(big_step(entry, true), bits - 128), power5));
|
|
}
|
|
power5 = big_mul(power5, {5});
|
|
}
|
|
|
|
power5 = {5};
|
|
for (std::int64_t q = -1; q >= nlohmann::detail::pow5_128_smallest_power; --q)
|
|
{
|
|
const auto index = static_cast<std::size_t>(2 * (q - nlohmann::detail::pow5_128_smallest_power));
|
|
const big_uint entry = big_from(table[index], table[index + 1]);
|
|
CHECK(big_bit_length(entry) == 128);
|
|
const std::size_t z = big_bit_length(power5);
|
|
const big_uint two_b = big_shl({1}, q >= -27 ? z + 127 : (2 * z) + 128);
|
|
// c = floor(2^b / p) + 1, stored as floor(c / 2^s):
|
|
// (entry * 2^s - 1) * p <= 2^b < ((entry + 1) * 2^s - 1) * p
|
|
const std::size_t s = q >= -27 ? 0 : z + 1;
|
|
CHECK(big_less_equal(big_mul(big_step(big_shl(entry, s), false), power5), two_b));
|
|
CHECK_FALSE(big_less_equal(big_mul(big_step(big_shl(big_step(entry, true), s), false), power5), two_b));
|
|
power5 = big_mul(power5, {5});
|
|
}
|
|
}
|
|
|
|
SECTION("128-bit products and leading zeros")
|
|
{
|
|
// whichever implementation the compiler gets (with or without a
|
|
// 128-bit integer type or a builtin)
|
|
std::uint64_t state = 42;
|
|
for (int i = 0; i < 10000; ++i)
|
|
{
|
|
state ^= state << 13u;
|
|
state ^= state >> 7u;
|
|
state ^= state << 17u;
|
|
const std::uint64_t a = state;
|
|
const std::uint64_t b = (state * 0x9E3779B97F4A7C15u) >> (i % 64);
|
|
const auto product = nlohmann::detail::full_multiplication(a, b);
|
|
CHECK(big_from(product.high, product.low) == big_mul(big_from(0, a), big_from(0, b)));
|
|
|
|
const int k = i % 64;
|
|
const std::uint64_t x = (std::uint64_t{1} << k) | (a & ((std::uint64_t{1} << k) - 1));
|
|
CHECK(nlohmann::detail::count_leading_zeros(x) == 63 - k);
|
|
}
|
|
}
|
|
|
|
SECTION("known values")
|
|
{
|
|
// Generated with Python, whose float() is correctly rounded:
|
|
// cases = [<hard cases>, 2**53 + 2k + 1, 2**54 + 4k + 2, and exact midpoints
|
|
// between neighbouring doubles, also 1e-60 above and below them]
|
|
// print('{"%s", 0x%016xu},' % (s, struct.unpack('<Q', struct.pack('<d', float(s)))[0]))
|
|
const std::vector<std::pair<std::string, std::uint64_t>> known =
|
|
{
|
|
{"0", 0x0000000000000000u},
|
|
{"-0", 0x8000000000000000u},
|
|
{"0.0", 0x0000000000000000u},
|
|
{"-0.0", 0x8000000000000000u},
|
|
{"0e5", 0x0000000000000000u},
|
|
{"0.000e-9", 0x0000000000000000u},
|
|
{"1", 0x3ff0000000000000u},
|
|
{"-1", 0xbff0000000000000u},
|
|
{"0.1", 0x3fb999999999999au},
|
|
{"0.3", 0x3fd3333333333333u},
|
|
{"1.5", 0x3ff8000000000000u},
|
|
{"-2.5e-3", 0xbf647ae147ae147bu},
|
|
{"1e23", 0x44b52d02c7e14af6u},
|
|
{"1e22", 0x4480f0cf064dd592u},
|
|
{"8.98846567431158e307", 0x7fe0000000000000u},
|
|
{"2.2250738585072011e-308", 0x000fffffffffffffu},
|
|
{"2.2250738585072012e-308", 0x0010000000000000u},
|
|
{"2.2250738585072014e-308", 0x0010000000000000u},
|
|
{"4.9406564584124654e-324", 0x0000000000000001u},
|
|
{"2.4703282292062327e-324", 0x0000000000000000u},
|
|
{"2.4703282292062328e-324", 0x0000000000000001u},
|
|
{"1e-324", 0x0000000000000000u},
|
|
{"3e-324", 0x0000000000000001u},
|
|
{"1.7976931348623157e308", 0x7fefffffffffffffu},
|
|
{"1.7976931348623158e308", 0x7fefffffffffffffu},
|
|
{"1.7976931348623159e308", 0x7ff0000000000000u},
|
|
{"1e308", 0x7fe1ccf385ebc8a0u},
|
|
{"1e309", 0x7ff0000000000000u},
|
|
{"-1e400", 0xfff0000000000000u},
|
|
{"1e-400", 0x0000000000000000u},
|
|
{"9007199254740991", 0x433fffffffffffffu},
|
|
{"9007199254740992", 0x4340000000000000u},
|
|
{"9007199254740993", 0x4340000000000000u},
|
|
{"9007199254740995", 0x4340000000000002u},
|
|
{"18014398509481986", 0x4350000000000000u},
|
|
{"18014398509481990", 0x4350000000000002u},
|
|
{"7.2057594037927933e16", 0x4370000000000000u},
|
|
{"123456789012345678901234567890", 0x45f8ee90ff6c373eu},
|
|
{"1.000000000000000111", 0x3ff0000000000000u},
|
|
{"1.0000000000000001110223", 0x3ff0000000000000u},
|
|
{"1.00000000000000011102230246251565404236316680908203125", 0x3ff0000000000000u},
|
|
{"1.00000000000000011102230246251565404236316680908203126", 0x3ff0000000000001u},
|
|
{"0.00000000000000000000000000000000000000000000000000000000000001", 0x3310747ddddf22a8u},
|
|
{"100000000000000000000000000000000000000000000", 0x4911efc659cf7d4cu},
|
|
{"1234567890123456789", 0x43b12210f47de981u},
|
|
{"12345678901234567890", 0x43e56a95319d63e1u},
|
|
{"1234567890123456789.5", 0x43b12210f47de981u},
|
|
{"0.1234567890123456789012345", 0x3fbf9add3746f65fu},
|
|
{"4.4501477170144023e-308", 0x001fffffffffffffu},
|
|
{"2.4406961166466664e-309", 0x0001c14ae5310a48u},
|
|
{"5e-324", 0x0000000000000001u},
|
|
{"1.0e-307", 0x0031fa182c40c60du},
|
|
{"179769313486231570814527423731704356798070567525844996598917476803157260780028538760589558632766878171540458953514382464234321326889464182768467546703537516986049910576551282076245490090389328944075868508455133942304583236903222948165808559332123348274797826204144723168738177180919299881250404026184124858368", 0x7fefffffffffffffu},
|
|
{"4.9e-324", 0x0000000000000001u},
|
|
{"9007199254740993", 0x4340000000000000u},
|
|
{"18014398509481986", 0x4350000000000000u},
|
|
{"9007199254740995", 0x4340000000000002u},
|
|
{"18014398509481990", 0x4350000000000002u},
|
|
{"9007199254740997", 0x4340000000000002u},
|
|
{"18014398509481994", 0x4350000000000002u},
|
|
{"9007199254740999", 0x4340000000000004u},
|
|
{"18014398509481998", 0x4350000000000004u},
|
|
{"9007199254741001", 0x4340000000000004u},
|
|
{"18014398509482002", 0x4350000000000004u},
|
|
{"9007199254741003", 0x4340000000000006u},
|
|
{"18014398509482006", 0x4350000000000006u},
|
|
{"9007199254741005", 0x4340000000000006u},
|
|
{"18014398509482010", 0x4350000000000006u},
|
|
{"9007199254741007", 0x4340000000000008u},
|
|
{"18014398509482014", 0x4350000000000008u},
|
|
{"9007199254741009", 0x4340000000000008u},
|
|
{"18014398509482018", 0x4350000000000008u},
|
|
{"9007199254741011", 0x434000000000000au},
|
|
{"18014398509482022", 0x435000000000000au},
|
|
{"9007199254741013", 0x434000000000000au},
|
|
{"18014398509482026", 0x435000000000000au},
|
|
{"9007199254741015", 0x434000000000000cu},
|
|
{"18014398509482030", 0x435000000000000cu},
|
|
{"9007199254741017", 0x434000000000000cu},
|
|
{"18014398509482034", 0x435000000000000cu},
|
|
{"9007199254741019", 0x434000000000000eu},
|
|
{"18014398509482038", 0x435000000000000eu},
|
|
{"9007199254741021", 0x434000000000000eu},
|
|
{"18014398509482042", 0x435000000000000eu},
|
|
{"9007199254741023", 0x4340000000000010u},
|
|
{"18014398509482046", 0x4350000000000010u},
|
|
{"9007199254741025", 0x4340000000000010u},
|
|
{"18014398509482050", 0x4350000000000010u},
|
|
{"9007199254741027", 0x4340000000000012u},
|
|
{"18014398509482054", 0x4350000000000012u},
|
|
{"9007199254741029", 0x4340000000000012u},
|
|
{"18014398509482058", 0x4350000000000012u},
|
|
{"9007199254741031", 0x4340000000000014u},
|
|
{"18014398509482062", 0x4350000000000014u},
|
|
{"9007199254741033", 0x4340000000000014u},
|
|
{"18014398509482066", 0x4350000000000014u},
|
|
{"9007199254741035", 0x4340000000000016u},
|
|
{"18014398509482070", 0x4350000000000016u},
|
|
{"9007199254741037", 0x4340000000000016u},
|
|
{"18014398509482074", 0x4350000000000016u},
|
|
{"9007199254741039", 0x4340000000000018u},
|
|
{"18014398509482078", 0x4350000000000018u},
|
|
{"9007199254741041", 0x4340000000000018u},
|
|
{"18014398509482082", 0x4350000000000018u},
|
|
{"9007199254741043", 0x434000000000001au},
|
|
{"18014398509482086", 0x435000000000001au},
|
|
{"9007199254741045", 0x434000000000001au},
|
|
{"18014398509482090", 0x435000000000001au},
|
|
{"9007199254741047", 0x434000000000001cu},
|
|
{"18014398509482094", 0x435000000000001cu},
|
|
{"9007199254741049", 0x434000000000001cu},
|
|
{"18014398509482098", 0x435000000000001cu},
|
|
{"9007199254741051", 0x434000000000001eu},
|
|
{"18014398509482102", 0x435000000000001eu},
|
|
{"9007199254741053", 0x434000000000001eu},
|
|
{"18014398509482106", 0x435000000000001eu},
|
|
{"9007199254741055", 0x4340000000000020u},
|
|
{"18014398509482110", 0x4350000000000020u},
|
|
{"9007199254741057", 0x4340000000000020u},
|
|
{"18014398509482114", 0x4350000000000020u},
|
|
{"9007199254741059", 0x4340000000000022u},
|
|
{"18014398509482118", 0x4350000000000022u},
|
|
{"9007199254741061", 0x4340000000000022u},
|
|
{"18014398509482122", 0x4350000000000022u},
|
|
{"9007199254741063", 0x4340000000000024u},
|
|
{"18014398509482126", 0x4350000000000024u},
|
|
{"9007199254741065", 0x4340000000000024u},
|
|
{"18014398509482130", 0x4350000000000024u},
|
|
{"9007199254741067", 0x4340000000000026u},
|
|
{"18014398509482134", 0x4350000000000026u},
|
|
{"9007199254741069", 0x4340000000000026u},
|
|
{"18014398509482138", 0x4350000000000026u},
|
|
{"9007199254741071", 0x4340000000000028u},
|
|
{"18014398509482142", 0x4350000000000028u},
|
|
{"0.00000000000000000142055942108419951085063380808124102279024543292671443374397544090470546507276594638824462890625", 0x3c3a3466f662d406u},
|
|
{"0.000000000000000001420559421084199510850633808081241022790245432926714433743975440904705465072765946388244628906251", 0x3c3a3466f662d407u},
|
|
{"0.00000000000000000142055942108419951085063380808124102279024543292671443374397444090470546507276594638824462890625", 0x3c3a3466f662d406u},
|
|
{"8656.5250079159513916238211095333099365234375", 0x40c0e84333759a94u},
|
|
{"8656.52500791595139162382110953330993652343751", 0x40c0e84333759a94u},
|
|
{"8656.525007915951391623821109533309936523437499999999999999999", 0x40c0e84333759a93u},
|
|
{"13.07696731650454946560557800694368779659271240234375", 0x402a276842967ef0u},
|
|
{"13.076967316504549465605578006943687796592712402343751", 0x402a276842967ef0u},
|
|
{"13.07696731650454946560557800694368779659271240234374999999999", 0x402a276842967eefu},
|
|
{"74708253715391928", 0x437096ac2cc7ee5cu},
|
|
{"747082537153919281", 0x43a4bc5737f9e9f2u},
|
|
{"74708253715391927.99999999999999999999999999999999999999999999", 0x437096ac2cc7ee5bu},
|
|
{"1809802988.27203977108001708984375", 0x41daf7d9bb11691au},
|
|
{"1809802988.272039771080017089843751", 0x41daf7d9bb11691au},
|
|
{"1809802988.272039771080017089843749999999999999999999999999999", 0x41daf7d9bb116919u},
|
|
{"51.390809684186766759239617385901510715484619140625", 0x4049b2060d3e4568u},
|
|
{"51.3908096841867667592396173859015107154846191406251", 0x4049b2060d3e4569u},
|
|
{"51.39080968418676675923961738590151071548461914062499999999999", 0x4049b2060d3e4568u},
|
|
{"9999807412.59738445281982421875", 0x4202a0479da4c772u},
|
|
{"9999807412.597384452819824218751", 0x4202a0479da4c772u},
|
|
{"9999807412.597384452819824218749999999999999999999999999999999", 0x4202a0479da4c771u},
|
|
{"0.00000000023260971767101600534534272272645127367651785021962496102787554264068603515625", 0x3deff83a135dec10u},
|
|
{"0.000000000232609717671016005345342722726451273676517850219624961027875542640686035156251", 0x3deff83a135dec11u},
|
|
{"0.00000000023260971767101600534534272272645127367651785021962496102787544264068603515625", 0x3deff83a135dec10u},
|
|
{"0.000000000497610482021202508216234789619066523902457532813059515319764614105224609375", 0x3e01190730d1ec48u},
|
|
{"0.0000000004976104820212025082162347896190665239024575328130595153197646141052246093751", 0x3e01190730d1ec48u},
|
|
{"0.000000000497610482021202508216234789619066523902457532813059515319764514105224609375", 0x3e01190730d1ec47u},
|
|
{"0.0000000000291336422596533830691676779231223432149733287843673679162748157978057861328125", 0x3dc004321559736eu},
|
|
{"0.00000000002913364225965338306916767792312234321497332878436736791627481579780578613281251", 0x3dc004321559736fu},
|
|
{"0.0000000000291336422596533830691676779231223432149733287843673679162748057978057861328125", 0x3dc004321559736eu},
|
|
{"0.000000000000000039237155154865396441907405399546892080260012902422940561653064150959835387766361236572265625", 0x3c869e61cfa3b8a4u},
|
|
{"0.0000000000000000392371551548653964419074053995468920802600129024229405616530641509598353877663612365722656251", 0x3c869e61cfa3b8a5u},
|
|
{"0.000000000000000039237155154865396441907405399546892080260012902422940561653054150959835387766361236572265625", 0x3c869e61cfa3b8a4u},
|
|
{"0.00000000000000006496592863767266414092845207857846208631635335985395063307379359685000963509082794189453125", 0x3c92b9a3b219ee84u},
|
|
{"0.000000000000000064965928637672664140928452078578462086316353359853950633073793596850009635090827941894531251", 0x3c92b9a3b219ee85u},
|
|
{"0.00000000000000006496592863767266414092845207857846208631635335985395063307378359685000963509082794189453125", 0x3c92b9a3b219ee84u},
|
|
{"0.0000000000448169607439179753541822628688597626549217078917308754171244800090789794921875", 0x3dc8a36d2e8094dau},
|
|
{"0.00000000004481696074391797535418226286885976265492170789173087541712448000907897949218751", 0x3dc8a36d2e8094dau},
|
|
{"0.0000000000448169607439179753541822628688597626549217078917308754171244700090789794921875", 0x3dc8a36d2e8094d9u},
|
|
{"1800873890234250.875", 0x4319978a820cfe2cu},
|
|
{"1800873890234250.8751", 0x4319978a820cfe2cu},
|
|
{"1800873890234250.874999999999999999999999999999999999999999999", 0x4319978a820cfe2bu},
|
|
{"0.000023941132739153309216405436654628857695570331998169422149658203125", 0x3ef91aa61d42e0e0u},
|
|
{"0.0000239411327391533092164054366546288576955703319981694221496582031251", 0x3ef91aa61d42e0e1u},
|
|
{"0.000023941132739153309216405436654628857695570331998169422149658193125", 0x3ef91aa61d42e0e0u},
|
|
{"8339818978785937.5", 0x433da1056bb8ba92u},
|
|
{"8339818978785937.51", 0x433da1056bb8ba92u},
|
|
{"8339818978785937.499999999999999999999999999999999999999999999", 0x433da1056bb8ba91u},
|
|
{"283649145986385328", 0x438f7dcb29dae42eu},
|
|
{"2836491459863853281", 0x43c3ae9efa28ce9cu},
|
|
{"283649145986385327.9999999999999999999999999999999999999999999", 0x438f7dcb29dae42du},
|
|
{"0.0000000000000000004203729478287971874304476341685497759528753070014375965192388040492232903488911688327789306640625", 0x3c1f049ed78cb8a2u},
|
|
{"0.00000000000000000042037294782879718743044763416854977595287530700143759651923880404922329034889116883277893066406251", 0x3c1f049ed78cb8a3u},
|
|
{"0.0000000000000000004203729478287971874304476341685497759528753070014375965192387040492232903488911688327789306640625", 0x3c1f049ed78cb8a2u},
|
|
{"340031.78635183119331486523151397705078125", 0x4114c0ff25396a18u},
|
|
{"340031.786351831193314865231513977050781251", 0x4114c0ff25396a19u},
|
|
{"340031.7863518311933148652315139770507812499999999999999999999", 0x4114c0ff25396a18u},
|
|
{"0.0000000004543709717853330820053712055931242029538363880192264332436025142669677734375", 0x3dff3960f070bf76u},
|
|
{"0.00000000045437097178533308200537120559312420295383638801922643324360251426696777343751", 0x3dff3960f070bf76u},
|
|
{"0.0000000004543709717853330820053712055931242029538363880192264332436024142669677734375", 0x3dff3960f070bf75u},
|
|
{"0.0000000000007937958898451257082591244100968761704503924570008877026339177973568439483642578125", 0x3d6bede0b40e37c2u},
|
|
{"0.00000000000079379588984512570825912441009687617045039245700088770263391779735684394836425781251", 0x3d6bede0b40e37c3u},
|
|
{"0.0000000000007937958898451257082591244100968761704503924570008877026339176973568439483642578125", 0x3d6bede0b40e37c2u},
|
|
{"0.00000000000000068304843500200357021582520000166071595321259251644419041582523277611471712589263916015625", 0x3cc89c02848f8654u},
|
|
{"0.000000000000000683048435002003570215825200001660715953212592516444190415825232776114717125892639160156251", 0x3cc89c02848f8655u},
|
|
{"0.00000000000000068304843500200357021582520000166071595321259251644419041582513277611471712589263916015625", 0x3cc89c02848f8654u},
|
|
{"0.00000000147705080029041786940146712084494760863773166192913777194917201995849609375", 0x3e1960235bc34d06u},
|
|
{"0.000000001477050800290417869401467120844947608637731661929137771949172019958496093751", 0x3e1960235bc34d06u},
|
|
{"0.00000000147705080029041786940146712084494760863773166192913777194917101995849609375", 0x3e1960235bc34d05u},
|
|
{"2164972979236447104", 0x43be0b87743fb524u},
|
|
{"21649729792364471041", 0x43f2c734a8a7d136u},
|
|
{"2164972979236447103.999999999999999999999999999999999999999999", 0x43be0b87743fb523u},
|
|
{"13124633159586767", 0x4347506464a469e8u},
|
|
{"131246331595867671", 0x437d247d7dcd8461u},
|
|
{"13124633159586766.99999999999999999999999999999999999999999999", 0x4347506464a469e7u},
|
|
{"0.000000000000027605165650764659887597286048864477738779108113853499872902830247767269611358642578125", 0x3d1f14a5b417cb08u},
|
|
{"0.0000000000000276051656507646598875972860488644777387791081138534998729028302477672696113586425781251", 0x3d1f14a5b417cb09u},
|
|
{"0.000000000000027605165650764659887597286048864477738779108113853499872902820247767269611358642578125", 0x3d1f14a5b417cb08u},
|
|
{"0.01727792833395616796388072344825559412129223346710205078125", 0x3f91b14e248c42c8u},
|
|
{"0.017277928333956167963880723448255594121292233467102050781251", 0x3f91b14e248c42c9u},
|
|
{"0.01727792833395616796388072344825559412129223346710205078124999", 0x3f91b14e248c42c8u},
|
|
{"0.00000000000003096211381890547702500862566064822950373937142376501441276559489779174327850341796875", 0x3d216e1c61130b22u},
|
|
{"0.000000000000030962113818905477025008625660648229503739371423765014412765594897791743278503417968751", 0x3d216e1c61130b22u},
|
|
{"0.00000000000003096211381890547702500862566064822950373937142376501441276558489779174327850341796875", 0x3d216e1c61130b21u},
|
|
{"0.0000000000000683414954481284270259359974108983284531919182025472281338807079009711742401123046875", 0x3d333c86137e5170u},
|
|
{"0.00000000000006834149544812842702593599741089832845319191820254722813388070790097117424011230468751", 0x3d333c86137e5170u},
|
|
{"0.0000000000000683414954481284270259359974108983284531919182025472281338806979009711742401123046875", 0x3d333c86137e516fu},
|
|
{"3237539054990129.75", 0x4327010c9aa36664u},
|
|
{"3237539054990129.751", 0x4327010c9aa36664u},
|
|
{"3237539054990129.749999999999999999999999999999999999999999999", 0x4327010c9aa36663u},
|
|
{"0.0000000000000105429301486355911969397173440055240907798901443814809653076736140064895153045654296875", 0x3d07bd95dfb8a1eeu},
|
|
{"0.00000000000001054293014863559119693971734400552409077989014438148096530767361400648951530456542968751", 0x3d07bd95dfb8a1eeu},
|
|
{"0.0000000000000105429301486355911969397173440055240907798901443814809653076636140064895153045654296875", 0x3d07bd95dfb8a1edu},
|
|
{"9460.2061893763575426419265568256378173828125", 0x40c27a1a6469da1eu},
|
|
{"9460.20618937635754264192655682563781738281251", 0x40c27a1a6469da1fu},
|
|
{"9460.206189376357542641926556825637817382812499999999999999999", 0x40c27a1a6469da1eu},
|
|
{"465000373656610.53125", 0x42fa6ea56179c228u},
|
|
{"465000373656610.531251", 0x42fa6ea56179c229u},
|
|
{"465000373656610.5312499999999999999999999999999999999999999999", 0x42fa6ea56179c228u},
|
|
{"0.000000000107709773707743713401260815383950956818093214195641849073581397533416748046875", 0x3ddd9b66c974bb14u},
|
|
{"0.0000000001077097737077437134012608153839509568180932141956418490735813975334167480468751", 0x3ddd9b66c974bb14u},
|
|
{"0.000000000107709773707743713401260815383950956818093214195641849073581297533416748046875", 0x3ddd9b66c974bb13u},
|
|
{"0.012083347821554271152300064073870089487172663211822509765625", 0x3f88bf277dc215f4u},
|
|
{"0.0120833478215542711523000640738700894871726632118225097656251", 0x3f88bf277dc215f5u},
|
|
{"0.01208334782155427115230006407387008948717266321182250976562499", 0x3f88bf277dc215f4u},
|
|
{"2309804058391724800", 0x43c0070946d098e2u},
|
|
{"23098040583917248001", 0x43f408cb9884bf1au},
|
|
{"2309804058391724799.999999999999999999999999999999999999999999", 0x43c0070946d098e1u},
|
|
{"0.000000000078286047058220682019445698291671103911937290575906445155851542949676513671875", 0x3dd584e40ca80638u},
|
|
{"0.0000000000782860470582206820194456982916711039119372905759064451558515429496765136718751", 0x3dd584e40ca80638u},
|
|
{"0.000000000078286047058220682019445698291671103911937290575906445155851532949676513671875", 0x3dd584e40ca80637u},
|
|
{"2940024994425.709228515625", 0x4285643929d3cdacu},
|
|
{"2940024994425.7092285156251", 0x4285643929d3cdadu},
|
|
{"2940024994425.709228515624999999999999999999999999999999999999", 0x4285643929d3cdacu},
|
|
{"9503358.427352792583405971527099609375", 0x4162204fcdacdfc4u},
|
|
{"9503358.4273527925834059715270996093751", 0x4162204fcdacdfc4u},
|
|
{"9503358.427352792583405971527099609374999999999999999999999999", 0x4162204fcdacdfc3u},
|
|
{"0.00005589147834433423614399101542193903924271580763161182403564453125", 0x3f0d4da092aa5d6au},
|
|
{"0.000055891478344334236143991015421939039242715807631611824035644531251", 0x3f0d4da092aa5d6bu},
|
|
{"0.00005589147834433423614399101542193903924271580763161182403564452125", 0x3f0d4da092aa5d6au},
|
|
{"0.0000000164797038506435664295103900420409737126448135313694365322589874267578125", 0x3e51b1e8107bd640u},
|
|
{"0.00000001647970385064356642951039004204097371264481353136943653225898742675781251", 0x3e51b1e8107bd641u},
|
|
{"0.0000000164797038506435664295103900420409737126448135313694365322589774267578125", 0x3e51b1e8107bd640u},
|
|
{"73055.7873927834807545877993106842041015625", 0x40f1d5fc99292ce2u},
|
|
{"73055.78739278348075458779931068420410156251", 0x40f1d5fc99292ce3u},
|
|
{"73055.78739278348075458779931068420410156249999999999999999999", 0x40f1d5fc99292ce2u},
|
|
{"0.00000000002558172364455232122427880575336067736115508441940846751094795763492584228515625", 0x3dbc209d7509115au},
|
|
{"0.000000000025581723644552321224278805753360677361155084419408467510947957634925842285156251", 0x3dbc209d7509115bu},
|
|
{"0.00000000002558172364455232122427880575336067736115508441940846751094794763492584228515625", 0x3dbc209d7509115au},
|
|
{"0.000000000854497507560657937804791333430312443020238077906469698064029216766357421875", 0x3e0d5c3d540cc0d2u},
|
|
{"0.0000000008544975075606579378047913334303124430202380779064696980640292167663574218751", 0x3e0d5c3d540cc0d2u},
|
|
{"0.000000000854497507560657937804791333430312443020238077906469698064029116766357421875", 0x3e0d5c3d540cc0d1u},
|
|
{"26671499731071461376", 0x43f722433b19970eu},
|
|
{"266714997310714613761", 0x442cead409dffcd2u},
|
|
{"26671499731071461375.99999999999999999999999999999999999999999", 0x43f722433b19970eu},
|
|
{"0.0000000002725195963972150049060368088050545186395989816219298518262803554534912109375", 0x3df2ba37271cf5f2u},
|
|
{"0.00000000027251959639721500490603680880505451863959898162192985182628035545349121093751", 0x3df2ba37271cf5f2u},
|
|
{"0.0000000002725195963972150049060368088050545186395989816219298518262802554534912109375", 0x3df2ba37271cf5f1u},
|
|
{"0.0000000000377224663702246439557904325637937886957218314165629635681398212909698486328125", 0x3dc4bcf7157af68eu},
|
|
{"0.00000000003772246637022464395579043256379378869572183141656296356813982129096984863281251", 0x3dc4bcf7157af68fu},
|
|
{"0.0000000000377224663702246439557904325637937886957218314165629635681398112909698486328125", 0x3dc4bcf7157af68eu},
|
|
{"0.00000000000000009977342762593364154535842258994910996905560208471673566688053824691451154649257659912109375", 0x3c9cc1fac312e2a6u},
|
|
{"0.000000000000000099773427625933641545358422589949109969055602084716735666880538246914511546492576599121093751", 0x3c9cc1fac312e2a6u},
|
|
{"0.00000000000000009977342762593364154535842258994910996905560208471673566688052824691451154649257659912109375", 0x3c9cc1fac312e2a5u},
|
|
{"0.0000000003146248171139977444431948290159907662133509376189977047033607959747314453125", 0x3df59ef03588a228u},
|
|
{"0.00000000031462481711399774444319482901599076621335093761899770470336079597473144531251", 0x3df59ef03588a229u},
|
|
{"0.0000000003146248171139977444431948290159907662133509376189977047033606959747314453125", 0x3df59ef03588a228u},
|
|
{"488899209263030304", 0x439b23acf64c5e80u},
|
|
{"4888992092630303041", 0x43d0f64c19efbb10u},
|
|
{"488899209263030303.9999999999999999999999999999999999999999999", 0x439b23acf64c5e80u},
|
|
{"1.88357157350592807620870416940306313335895538330078125", 0x3ffe231bf23e21acu},
|
|
{"1.883571573505928076208704169403063133358955383300781251", 0x3ffe231bf23e21adu},
|
|
{"1.883571573505928076208704169403063133358955383300781249999999", 0x3ffe231bf23e21acu},
|
|
{"0.0000000216458400594294836451424756990254139044083103726734407246112823486328125", 0x3e573df694e72fb8u},
|
|
{"0.00000002164584005942948364514247569902541390440831037267344072461128234863281251", 0x3e573df694e72fb9u},
|
|
{"0.0000000216458400594294836451424756990254139044083103726734407246112723486328125", 0x3e573df694e72fb8u},
|
|
{"5107.79271041116453488939441740512847900390625", 0x40b3f3caef11cb26u},
|
|
{"5107.792710411164534889394417405128479003906251", 0x40b3f3caef11cb27u},
|
|
{"5107.792710411164534889394417405128479003906249999999999999999", 0x40b3f3caef11cb26u},
|
|
{"734059.8035226609208621084690093994140625", 0x412666d79b67527cu},
|
|
{"734059.80352266092086210846900939941406251", 0x412666d79b67527du},
|
|
{"734059.8035226609208621084690093994140624999999999999999999999", 0x412666d79b67527cu},
|
|
{"61431562016722684", 0x436b47f5c40021e0u},
|
|
{"614315620167226841", 0x43a10cf99a80152cu},
|
|
{"61431562016722683.99999999999999999999999999999999999999999999", 0x436b47f5c40021dfu},
|
|
{"2.0060840449445034305853141631814651191234588623046875", 0x40000c75cab08326u},
|
|
{"2.00608404494450343058531416318146511912345886230468751", 0x40000c75cab08326u},
|
|
{"2.006084044944503430585314163181465119123458862304687499999999", 0x40000c75cab08325u},
|
|
{"0.0000001760623599453036952716420489480075861621344301966018974781036376953125", 0x3e87a174e55262cau},
|
|
{"0.00000017606235994530369527164204894800758616213443019660189747810363769531251", 0x3e87a174e55262cbu},
|
|
{"0.0000001760623599453036952716420489480075861621344301966018974781035376953125", 0x3e87a174e55262cau},
|
|
{"0.833085849636964581588216560703585855662822723388671875", 0x3feaa8a3a7de6fb6u},
|
|
{"0.8330858496369645815882165607035858556628227233886718751", 0x3feaa8a3a7de6fb6u},
|
|
{"0.8330858496369645815882165607035858556628227233886718749999999", 0x3feaa8a3a7de6fb5u},
|
|
{"45031428.4182307310402393341064453125", 0x418579002358895au},
|
|
{"45031428.41823073104023933410644531251", 0x418579002358895bu},
|
|
{"45031428.41823073104023933410644531249999999999999999999999999", 0x418579002358895au},
|
|
{"5003361733758455296", 0x43d15be0bf39dd24u},
|
|
{"50033617337584552961", 0x4405b2d8ef08546cu},
|
|
{"5003361733758455295.999999999999999999999999999999999999999999", 0x43d15be0bf39dd23u},
|
|
};
|
|
|
|
for (const auto& c : known)
|
|
{
|
|
CAPTURE(c.first);
|
|
CHECK(native_bits64(c.first) == c.second);
|
|
}
|
|
}
|
|
|
|
SECTION("binary32")
|
|
{
|
|
using binary32 = nlohmann::detail::ieee_binary_format<24>;
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(0, 1) == 0x3F800000u);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(-1, 1) == 0x3DCCCCCDu);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(-1, 15) == 0x3FC00000u);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(0, 16777217) == 0x4B800000u); // tie, to even
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(0, 16777219) == 0x4B800002u); // tie, to even
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(-45, 1) == 0x00000001u);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(-46, 7) == 0x00000000u);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(-46, 8) == 0x00000001u);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(-65, 9999999999999999999u) == 0x00000000u);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(20, 3402823466385288598u) == 0x7F7FFFFFu);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(20, 3402823669209384635u) == 0x7F800000u);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(39, 1) == 0x7F800000u);
|
|
CHECK(nlohmann::detail::eisel_lemire<binary32>(-5, 0) == 0x00000000u);
|
|
}
|
|
|
|
SECTION("round trip")
|
|
{
|
|
// every double written by to_chars and read back, and its 17-digit
|
|
// form with trailing digits that make the token longer than 19 digits
|
|
std::uint64_t state = 5295;
|
|
for (int i = 0; i < 200000; ++i)
|
|
{
|
|
state ^= state << 13u;
|
|
state ^= state >> 7u;
|
|
state ^= state << 17u;
|
|
std::uint64_t b = state;
|
|
if ((b & 0x7FF0000000000000u) == 0x7FF0000000000000u)
|
|
{
|
|
continue; // infinity or NaN
|
|
}
|
|
if (i % 4 == 0)
|
|
{
|
|
b &= 0x800FFFFFFFFFFFFFu; // subnormals
|
|
}
|
|
double d = 0;
|
|
std::memcpy(&d, &b, sizeof(d));
|
|
|
|
std::array<char, 64> buffer{};
|
|
const char* end = nlohmann::detail::to_chars(buffer.data(), buffer.data() + buffer.size(), d);
|
|
const std::string token(buffer.data(), static_cast<std::size_t>(end - buffer.data()));
|
|
CAPTURE(token);
|
|
CHECK(native_bits64(token) == b);
|
|
|
|
// insert digits before the exponent of the 17-digit form: that
|
|
// form lies strictly inside the rounding interval of the double
|
|
// (the shortest one may lie on its boundary), and the digits move
|
|
// it by far less than the distance to the boundary, so the value
|
|
// must not change
|
|
std::array<char, 64> digits17{};
|
|
static_cast<void>(std::snprintf(digits17.data(), digits17.size(), "%.17g", d)); // NOLINT(cppcoreguidelines-pro-type-vararg,hicpp-vararg)
|
|
std::string longer = digits17.data();
|
|
const std::size_t e = longer.find('e');
|
|
const std::size_t dot = longer.find('.');
|
|
const std::string extra = dot == std::string::npos ? ".000000000000000000001" : "000000000000000000001";
|
|
longer.insert(e == std::string::npos ? longer.size() : e, extra);
|
|
CAPTURE(longer);
|
|
CHECK(native_bits64(longer) == b);
|
|
}
|
|
}
|
|
|
|
SECTION("round trip, binary32")
|
|
{
|
|
std::uint32_t state = 5295;
|
|
for (int i = 0; i < 100000; ++i)
|
|
{
|
|
state ^= state << 13u;
|
|
state ^= state >> 17u;
|
|
state ^= state << 5u;
|
|
std::uint32_t b = state;
|
|
if ((b & 0x7F800000u) == 0x7F800000u)
|
|
{
|
|
continue; // infinity or NaN
|
|
}
|
|
if (i % 4 == 0)
|
|
{
|
|
b &= 0x807FFFFFu; // subnormals
|
|
}
|
|
float f = 0;
|
|
std::memcpy(&f, &b, sizeof(f));
|
|
|
|
std::array<char, 64> buffer{};
|
|
const char* end = nlohmann::detail::to_chars(buffer.data(), buffer.data() + buffer.size(), f);
|
|
const std::string token(buffer.data(), static_cast<std::size_t>(end - buffer.data()));
|
|
CAPTURE(token);
|
|
CHECK(native_bits32(token) == b);
|
|
}
|
|
}
|
|
|
|
SECTION("used by the lexer")
|
|
{
|
|
// 17 significant digits: beyond Clinger's fast path
|
|
CHECK(bits_of(json::parse("-65.613616999999977").get<double>()) == bits_of(-65.613616999999977));
|
|
CHECK(bits_of(json::parse("2.2250738585072011e-308").get<double>()) == 0x000FFFFFFFFFFFFFu);
|
|
CHECK(bits_of(json::parse("4.9406564584124654e-324").get<double>()) == 1u);
|
|
json _;
|
|
CHECK_THROWS_WITH_AS(_ = json::parse("1.7976931348623159e308"),
|
|
"[json.exception.out_of_range.406] number overflow parsing '1.7976931348623159e308'", json::out_of_range&);
|
|
}
|
|
}
|
|
|
|
namespace
|
|
{
|
|
using float_json = nlohmann::basic_json<std::map, std::vector, std::string, bool, std::int64_t, std::uint64_t, float>;
|
|
|
|
// the bits of the float that parse() gives for a token, via both scanners;
|
|
// the value must be the same for both
|
|
template<typename Json, typename Bits>
|
|
void check_parse(const std::string& token, Bits expected, Bits infinity)
|
|
{
|
|
std::stringstream stream(token);
|
|
if ((expected & ~(Bits{1} << (8 * sizeof(Bits) - 1))) == infinity)
|
|
{
|
|
Json _;
|
|
CHECK_THROWS_WITH_AS(_ = Json::parse(token), ("[json.exception.out_of_range.406] number overflow parsing '" + token + "'").c_str(), typename Json::out_of_range&);
|
|
CHECK_THROWS_WITH_AS(_ = Json::parse(stream), ("[json.exception.out_of_range.406] number overflow parsing '" + token + "'").c_str(), typename Json::out_of_range&);
|
|
return;
|
|
}
|
|
const Json contiguous = Json::parse(token);
|
|
const Json streamed = Json::parse(stream);
|
|
if (contiguous.is_number_float()) // not an integer that fits
|
|
{
|
|
CHECK(bits_of(contiguous.template get<typename Json::number_float_t>()) == expected);
|
|
CHECK(bits_of(streamed.template get<typename Json::number_float_t>()) == expected);
|
|
}
|
|
else
|
|
{
|
|
CHECK(streamed.is_number_integer());
|
|
}
|
|
}
|
|
} // namespace
|
|
|
|
TEST_CASE("float conversion of hard cases")
|
|
{
|
|
// see float_hard_cases.hpp
|
|
for (const auto& c : float_hard_cases::cases())
|
|
{
|
|
const std::string token = c.token;
|
|
CAPTURE(token);
|
|
CHECK(native_bits64(token) == c.bits64);
|
|
CHECK(native_bits32(token) == c.bits32);
|
|
check_parse<json>(token, c.bits64, std::uint64_t{0x7FF0000000000000u});
|
|
check_parse<float_json>(token, c.bits32, std::uint32_t{0x7F800000u});
|
|
}
|
|
}
|
|
|
|
TEST_CASE("float overflow and underflow in the parser")
|
|
{
|
|
SECTION("double")
|
|
{
|
|
check_parse<json>("1.7976931348623157e308", std::uint64_t{0x7FEFFFFFFFFFFFFFu}, std::uint64_t{0x7FF0000000000000u});
|
|
check_parse<json>("1.7976931348623159e308", std::uint64_t{0x7FF0000000000000u}, std::uint64_t{0x7FF0000000000000u});
|
|
check_parse<json>("-1e309", std::uint64_t{0xFFF0000000000000u}, std::uint64_t{0x7FF0000000000000u});
|
|
check_parse<json>("1" + std::string(400, '0'), std::uint64_t{0x7FF0000000000000u}, std::uint64_t{0x7FF0000000000000u});
|
|
check_parse<json>("1e99999999999999999999", std::uint64_t{0x7FF0000000000000u}, std::uint64_t{0x7FF0000000000000u});
|
|
// an underflow gives a zero with the sign of the token
|
|
check_parse<json>("1e-400", std::uint64_t{0}, std::uint64_t{0x7FF0000000000000u});
|
|
check_parse<json>("-1e-400", std::uint64_t{0x8000000000000000u}, std::uint64_t{0x7FF0000000000000u});
|
|
check_parse<json>("-2.4703282292062327e-324", std::uint64_t{0x8000000000000000u}, std::uint64_t{0x7FF0000000000000u});
|
|
check_parse<json>("0." + std::string(400, '0') + "1", std::uint64_t{0}, std::uint64_t{0x7FF0000000000000u});
|
|
}
|
|
|
|
SECTION("float")
|
|
{
|
|
check_parse<float_json>("3.4028234e38", std::uint32_t{0x7F7FFFFFu}, std::uint32_t{0x7F800000u});
|
|
check_parse<float_json>("3.4028236e38", std::uint32_t{0x7F800000u}, std::uint32_t{0x7F800000u});
|
|
check_parse<float_json>("-1e39", std::uint32_t{0xFF800000u}, std::uint32_t{0x7F800000u});
|
|
check_parse<float_json>("1e-46", std::uint32_t{0}, std::uint32_t{0x7F800000u});
|
|
check_parse<float_json>("-1e-46", std::uint32_t{0x80000000u}, std::uint32_t{0x7F800000u});
|
|
check_parse<float_json>("-7.006492321624085e-46", std::uint32_t{0x80000000u}, std::uint32_t{0x7F800000u});
|
|
check_parse<float_json>("-7.006492321624086e-46", std::uint32_t{0x80000001u}, std::uint32_t{0x7F800000u});
|
|
}
|
|
}
|
|
|
|
TEST_CASE("string scanning kernels")
|
|
{
|
|
// the word-at-a-time kernels must stop exactly where a byte-by-byte scan
|
|
// stops, for any content, length, and alignment
|
|
const auto reference_special = [](const unsigned char* data, std::size_t n)
|
|
{
|
|
std::size_t i = 0;
|
|
while (i < n && !nlohmann::detail::is_string_special(data[i]))
|
|
{
|
|
++i;
|
|
}
|
|
return i;
|
|
};
|
|
const auto reference_copyable = [](const unsigned char* data, std::size_t n)
|
|
{
|
|
std::size_t i = 0;
|
|
while (i < n && nlohmann::detail::is_ascii_copyable(data[i]))
|
|
{
|
|
++i;
|
|
}
|
|
return i;
|
|
};
|
|
const auto reference_bulk_run = [](const unsigned char* data, std::size_t n)
|
|
{
|
|
std::size_t i = 0;
|
|
while (i < n)
|
|
{
|
|
if (data[i] < 0x80u)
|
|
{
|
|
if (nlohmann::detail::is_string_special(data[i]))
|
|
{
|
|
break;
|
|
}
|
|
++i;
|
|
continue;
|
|
}
|
|
const std::size_t seq = nlohmann::detail::validate_one_utf8(data + i, n - i);
|
|
if (seq == 0)
|
|
{
|
|
break;
|
|
}
|
|
i += seq;
|
|
}
|
|
return i;
|
|
};
|
|
|
|
// pieces: ordinary ASCII, stops, DEL, well-formed sequences of every
|
|
// length, and ill-formed or truncated ones
|
|
const std::vector<std::string> pieces =
|
|
{
|
|
"a", "Z", " ", "~", "0123456789", "\"", "\\", std::string(1, '\0'), "\n", "\x1F", "\x7F",
|
|
"\xC3\xA4", "\xE2\x82\xAC", "\xE6\x97\xA5\xE6\x9C\xAC", "\xF0\x9F\x98\x80", "\xED\x9F\xBF",
|
|
"\x80", "\xC0\x80", "\xC3", "\xE2\x82", "\xED\xA0\x80", "\xF4\x90\x80\x80", "\xFF",
|
|
};
|
|
std::uint64_t state = 5295;
|
|
const auto next = [&state]()
|
|
{
|
|
state ^= state << 13u;
|
|
state ^= state >> 7u;
|
|
state ^= state << 17u;
|
|
return state;
|
|
};
|
|
// the upper half as a 32-bit value: converts to std::size_t implicitly on
|
|
// every platform (a cast of std::uint64_t is useless where both are the
|
|
// same type, and required where std::size_t is 32 bits wide)
|
|
const auto next_small = [&next]()
|
|
{
|
|
return static_cast<std::uint32_t>(next() >> 32u);
|
|
};
|
|
for (int round = 0; round < 100000; ++round)
|
|
{
|
|
// mostly ordinary text, so that runs span several words
|
|
std::string text(next_small() % 8u, '.');
|
|
const std::size_t count = next_small() % 12u;
|
|
for (std::size_t k = 0; k < count; ++k)
|
|
{
|
|
const std::size_t p = (next() % 4 == 0) ? next_small() % pieces.size() : 0;
|
|
text += pieces[p];
|
|
text += std::string(next_small() % 10u, 'x');
|
|
}
|
|
const auto* data = reinterpret_cast<const unsigned char*>(text.data()); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
|
for (std::size_t offset = 0; offset < 3 && offset <= text.size(); ++offset)
|
|
{
|
|
const std::size_t n = text.size() - offset;
|
|
CAPTURE(text);
|
|
CAPTURE(offset);
|
|
CHECK(nlohmann::detail::find_string_special(data + offset, n) == reference_special(data + offset, n));
|
|
CHECK(nlohmann::detail::find_ascii_copyable_run(data + offset, n) == reference_copyable(data + offset, n));
|
|
CHECK(nlohmann::detail::scalar_string_bulk_run(data + offset, n) == reference_bulk_run(data + offset, n));
|
|
}
|
|
}
|
|
|
|
// the trailing-zero count, whichever implementation the compiler gets
|
|
for (int k = 0; k < 64; ++k)
|
|
{
|
|
const std::uint64_t bit = std::uint64_t{1} << k;
|
|
CHECK(nlohmann::detail::count_trailing_zeros(bit) == k);
|
|
CHECK(nlohmann::detail::count_trailing_zeros(bit | (bit << 1u) | 0x8000000000000000u) == k);
|
|
}
|
|
}
|