mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 22:15:19 +00:00
Find the stop byte of a string run without a byte loop
find_string_special() and find_ascii_copyable_run() test eight bytes at a time, but located the stopping byte inside a word with a byte loop. The lowest flagged byte of the SWAR tests is always a true hit (the borrows of the subtractions can only flag bytes above one), so its index is now the trailing-zero count of the mask; words are read in little-endian order on every platform, so this does not depend on the byte order. scalar_string_bulk_run() validates a run of multi-byte UTF-8 sequences one after another instead of searching for the next special byte in between, which helps text in non-Latin scripts. The kernels serve the lexer's contiguous fast path, the serializer, and the binary formats. New tests compare all three with byte-by-byte reference scans on 100,000 generated buffers at three alignments; the portable fallback of count_trailing_zeros() was checked against the builtin. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -1313,3 +1313,97 @@ TEST_CASE("Eisel-Lemire float conversion")
|
||||
"[json.exception.out_of_range.406] number overflow parsing '1.7976931348623159e308'", json::out_of_range&);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_CASE("string scanning kernels")
|
||||
{
|
||||
// the word-at-a-time kernels must stop exactly where a byte-by-byte scan
|
||||
// stops, for any content, length, and alignment
|
||||
const auto reference_special = [](const unsigned char* data, std::size_t n)
|
||||
{
|
||||
std::size_t i = 0;
|
||||
while (i < n && !nlohmann::detail::is_string_special(data[i]))
|
||||
{
|
||||
++i;
|
||||
}
|
||||
return i;
|
||||
};
|
||||
const auto reference_copyable = [](const unsigned char* data, std::size_t n)
|
||||
{
|
||||
std::size_t i = 0;
|
||||
while (i < n && nlohmann::detail::is_ascii_copyable(data[i]))
|
||||
{
|
||||
++i;
|
||||
}
|
||||
return i;
|
||||
};
|
||||
const auto reference_bulk_run = [](const unsigned char* data, std::size_t n)
|
||||
{
|
||||
std::size_t i = 0;
|
||||
while (i < n)
|
||||
{
|
||||
if (data[i] < 0x80u)
|
||||
{
|
||||
if (nlohmann::detail::is_string_special(data[i]))
|
||||
{
|
||||
break;
|
||||
}
|
||||
++i;
|
||||
continue;
|
||||
}
|
||||
const std::size_t seq = nlohmann::detail::validate_one_utf8(data + i, n - i);
|
||||
if (seq == 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
i += seq;
|
||||
}
|
||||
return i;
|
||||
};
|
||||
|
||||
// pieces: ordinary ASCII, stops, DEL, well-formed sequences of every
|
||||
// length, and ill-formed or truncated ones
|
||||
const std::vector<std::string> pieces =
|
||||
{
|
||||
"a", "Z", " ", "~", "0123456789", "\"", "\\", std::string(1, '\0'), "\n", "\x1F", "\x7F",
|
||||
"\xC3\xA4", "\xE2\x82\xAC", "\xE6\x97\xA5\xE6\x9C\xAC", "\xF0\x9F\x98\x80", "\xED\x9F\xBF",
|
||||
"\x80", "\xC0\x80", "\xC3", "\xE2\x82", "\xED\xA0\x80", "\xF4\x90\x80\x80", "\xFF",
|
||||
};
|
||||
std::uint64_t state = 5295;
|
||||
const auto next = [&state]()
|
||||
{
|
||||
state ^= state << 13u;
|
||||
state ^= state >> 7u;
|
||||
state ^= state << 17u;
|
||||
return state;
|
||||
};
|
||||
for (int round = 0; round < 100000; ++round)
|
||||
{
|
||||
// mostly ordinary text, so that runs span several words
|
||||
std::string text(static_cast<std::size_t>(next() % 8), '.');
|
||||
const auto count = static_cast<std::size_t>(next() % 12);
|
||||
for (std::size_t k = 0; k < count; ++k)
|
||||
{
|
||||
const std::size_t p = (next() % 4 == 0) ? static_cast<std::size_t>(next() % pieces.size()) : 0;
|
||||
text += pieces[p];
|
||||
text += std::string(static_cast<std::size_t>(next() % 10), 'x');
|
||||
}
|
||||
const auto* data = reinterpret_cast<const unsigned char*>(text.data()); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast)
|
||||
for (std::size_t offset = 0; offset < 3 && offset <= text.size(); ++offset)
|
||||
{
|
||||
const std::size_t n = text.size() - offset;
|
||||
CAPTURE(text);
|
||||
CAPTURE(offset);
|
||||
CHECK(nlohmann::detail::find_string_special(data + offset, n) == reference_special(data + offset, n));
|
||||
CHECK(nlohmann::detail::find_ascii_copyable_run(data + offset, n) == reference_copyable(data + offset, n));
|
||||
CHECK(nlohmann::detail::scalar_string_bulk_run(data + offset, n) == reference_bulk_run(data + offset, n));
|
||||
}
|
||||
}
|
||||
|
||||
// the trailing-zero count, whichever implementation the compiler gets
|
||||
for (int k = 0; k < 64; ++k)
|
||||
{
|
||||
const std::uint64_t bit = std::uint64_t{1} << k;
|
||||
CHECK(nlohmann::detail::count_trailing_zeros(bit) == k);
|
||||
CHECK(nlohmann::detail::count_trailing_zeros(bit | (bit << 1u) | 0x8000000000000000u) == k);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user