mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 22:15:19 +00:00
get_codepoint() read the four hex digits of a \u escape with four calls to get(), each classified by a chain of range comparisons. For contiguous input, get_codepoint_bulk() now decodes them with one lookup per byte (hex_codepoint() in string_scan.hpp, after yyjson's read_hex_u16): a 256-entry table maps a byte to its value, or 0xFF for anything else, and an invalid digit shows in the OR of the four values. It then skips the four bytes and updates the position counters as four get() calls would. If a digit is invalid or fewer than four bytes are left, it changes nothing and the existing loop runs, so errors are reported with the same message and position as before. json::parse, best of 5 runs in separate processes (M1 Max): the escaped twitter.json (every non-ASCII character as \u) -13.6%, all other files within 0.3%. Tests compare the contiguous and the streaming path (value or exception message) for valid escapes, surrogate pairs, truncated and invalid digits at every position, and 3,000 seeded random escapes. Signed-off-by: Niels Lohmann <mail@nlohmann.me>
366 lines
15 KiB
C++
366 lines
15 KiB
C++
// __ _____ _____ _____
|
|
// __| | __| | | | JSON for Modern C++
|
|
// | | |__ | | | | | | version 3.12.0
|
|
// |_____|_____|_____|_|___| https://github.com/nlohmann/json
|
|
//
|
|
// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me>
|
|
// SPDX-License-Identifier: MIT
|
|
|
|
#pragma once
|
|
|
|
#include <array> // array
|
|
#include <cstddef> // size_t
|
|
#include <cstdint> // uint64_t, uint8_t
|
|
#include <cstring> // memcpy
|
|
|
|
#include <nlohmann/detail/bit_ops.hpp>
|
|
#include <nlohmann/detail/macro_scope.hpp>
|
|
|
|
// Optional SIMD backend for bulk UTF-8 validation. This is an opt-in external
|
|
// dependency: nlohmann/json itself stays header-only and the C++11 scalar
|
|
// validator below is always available; defining JSON_USE_SIMDUTF additionally
|
|
// requires the simdutf headers on the include path and linking the simdutf
|
|
// library. See string_bulk_run().
|
|
//
|
|
// simdutf.h itself requires C++17 - it rejects older standards with an #error -
|
|
// so the backend is only compiled in from C++17 on. Below that the macro has no
|
|
// effect and the scalar validator is used; it accepts and rejects exactly the
|
|
// same input, so only throughput differs. macro_scope.hpp is included above to
|
|
// have JSON_HAS_CPP_17 available for this test.
|
|
#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17)
|
|
#include <simdutf.h>
|
|
#endif
|
|
|
|
// This file contains the byte-level string-scanning helpers used by the lexer's
|
|
// contiguous fast path. They operate purely on raw bytes (no dependency on the
|
|
// lexer's template parameters) so they are free functions, keeping the lexer
|
|
// itself focused on the state machine; see lexer::scan_string_bulk().
|
|
|
|
NLOHMANN_JSON_NAMESPACE_BEGIN
|
|
namespace detail
|
|
{
|
|
|
|
// classify a single byte as needing individual string handling: the closing
|
|
// quote, an escape, a control character, or a non-ASCII (UTF-8)
|
|
// lead/continuation byte. Ordinary bytes (0x20..0x7F except '"' and '\\') are
|
|
// copied verbatim, which the bulk scanner does 8 bytes at a time.
|
|
inline bool is_string_special(unsigned char c) noexcept
|
|
{
|
|
return c == '\"' || c == '\\' || c < 0x20u || c >= 0x80u;
|
|
}
|
|
|
|
// SWAR helper: return a word whose high bit is set in every byte of @a v that
|
|
// is_string_special(); zero if the 8 bytes are all ordinary.
|
|
inline std::uint64_t swar_string_special(std::uint64_t v) noexcept
|
|
{
|
|
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
|
constexpr std::uint64_t high = 0x8080808080808080ull;
|
|
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
|
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
|
const std::uint64_t has_quote = (q - ones) & ~q & high;
|
|
const std::uint64_t has_backslash = (b - ones) & ~b & high;
|
|
const std::uint64_t has_control = (v - 0x2020202020202020ull) & ~v & high; // < 0x20
|
|
const std::uint64_t has_non_ascii = v & high; // >= 0x80
|
|
return has_quote | has_backslash | has_control | has_non_ascii;
|
|
}
|
|
|
|
// return the index of the first is_string_special() byte in [data, data+n), or
|
|
// n if every byte is ordinary; scans 8 bytes at a time
|
|
inline std::size_t find_string_special(const unsigned char* data, std::size_t n) noexcept
|
|
{
|
|
std::size_t i = 0;
|
|
for (; i + 8 <= n; i += 8)
|
|
{
|
|
const std::uint64_t special = swar_string_special(read_eight_bytes(data + i));
|
|
if (special != 0)
|
|
{
|
|
// the lowest flagged byte is the first special one: the borrows of
|
|
// the subtractions can only flag bytes above a true hit
|
|
return i + (static_cast<std::size_t>(count_trailing_zeros(special)) / 8);
|
|
}
|
|
}
|
|
for (; i < n; ++i)
|
|
{
|
|
if (is_string_special(data[i]))
|
|
{
|
|
return i;
|
|
}
|
|
}
|
|
return n;
|
|
}
|
|
|
|
// classify a byte as one the serializer must NOT copy verbatim when
|
|
// ensure_ascii is requested: the closing quote, an escape, a control character
|
|
// (< 0x20), DEL (0x7F), or any non-ASCII byte (>= 0x80). Everything else -
|
|
// printable ASCII except '"' and '\\' - is emitted unchanged. Note this differs
|
|
// from is_string_special() only in that 0x7F is also a stop (it is escaped as
|
|
// \u007f under ensure_ascii).
|
|
inline bool is_ascii_copyable(unsigned char c) noexcept
|
|
{
|
|
return c >= 0x20u && c < 0x7Fu && c != '"' && c != '\\';
|
|
}
|
|
|
|
// return the index of the first byte in [data, data+n) that is NOT
|
|
// is_ascii_copyable(), or n if every byte can be copied verbatim; scans 8 bytes
|
|
// at a time. Used by the serializer's ensure_ascii fast path.
|
|
inline std::size_t find_ascii_copyable_run(const unsigned char* data, std::size_t n) noexcept
|
|
{
|
|
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
|
constexpr std::uint64_t high = 0x8080808080808080ull;
|
|
std::size_t i = 0;
|
|
for (; i + 8 <= n; i += 8)
|
|
{
|
|
const std::uint64_t v = read_eight_bytes(data + i);
|
|
const std::uint64_t q = v ^ 0x2222222222222222ull; // '"' (0x22)
|
|
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull; // '\\' (0x5C)
|
|
const std::uint64_t d = v ^ 0x7F7F7F7F7F7F7F7Full; // DEL (0x7F)
|
|
const std::uint64_t stop = ((q - ones) & ~q & high) // == '"'
|
|
| ((b - ones) & ~b & high) // == '\\'
|
|
| ((d - ones) & ~d & high) // == 0x7F
|
|
| ((v - 0x2020202020202020ull) & ~v & high) // < 0x20
|
|
| (v & high); // >= 0x80
|
|
if (stop != 0)
|
|
{
|
|
// the lowest flagged byte is the first one to stop at (see
|
|
// find_string_special())
|
|
return i + (static_cast<std::size_t>(count_trailing_zeros(stop)) / 8);
|
|
}
|
|
}
|
|
for (; i < n; ++i)
|
|
{
|
|
if (!is_ascii_copyable(data[i]))
|
|
{
|
|
return i;
|
|
}
|
|
}
|
|
return n;
|
|
}
|
|
|
|
// Validate one UTF-8 sequence at the front of [data, data+avail). Returns its
|
|
// length (2..4) only when the bytes form a *well-formed* sequence using exactly
|
|
// the same ranges as scan_string()'s per-byte switch, so the bulk path accepts
|
|
// precisely what the byte path accepts. Returns 0 for anything that is invalid,
|
|
// incomplete, or that the byte path must diagnose (the caller then defers to
|
|
// that path, keeping error messages unchanged). Lead bytes < 0x80 are handled
|
|
// by the caller and never passed here.
|
|
inline std::size_t validate_one_utf8(const unsigned char* data, std::size_t avail) noexcept
|
|
{
|
|
const unsigned char c0 = data[0];
|
|
if (c0 >= 0xC2 && c0 <= 0xDF) // U+0080..U+07FF
|
|
{
|
|
if (avail >= 2 && data[1] >= 0x80 && data[1] <= 0xBF)
|
|
{
|
|
return 2;
|
|
}
|
|
}
|
|
else if (c0 == 0xE0) // U+0800..U+0FFF
|
|
{
|
|
if (avail >= 3 && data[1] >= 0xA0 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
|
{
|
|
return 3;
|
|
}
|
|
}
|
|
else if ((c0 >= 0xE1 && c0 <= 0xEC) || c0 == 0xEE || c0 == 0xEF) // U+1000..U+CFFF, U+E000..U+FFFF
|
|
{
|
|
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF)
|
|
{
|
|
return 3;
|
|
}
|
|
}
|
|
else if (c0 == 0xED) // U+D000..U+D7FF (excludes surrogates)
|
|
{
|
|
if (avail >= 3 && data[1] >= 0x80 && data[1] <= 0x9F && data[2] >= 0x80 && data[2] <= 0xBF)
|
|
{
|
|
return 3;
|
|
}
|
|
}
|
|
else if (c0 == 0xF0) // U+10000..U+3FFFF
|
|
{
|
|
if (avail >= 4 && data[1] >= 0x90 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
|
{
|
|
return 4;
|
|
}
|
|
}
|
|
else if (c0 >= 0xF1 && c0 <= 0xF3) // U+40000..U+FFFFF
|
|
{
|
|
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0xBF && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
|
{
|
|
return 4;
|
|
}
|
|
}
|
|
else if (c0 == 0xF4) // U+100000..U+10FFFF
|
|
{
|
|
if (avail >= 4 && data[1] >= 0x80 && data[1] <= 0x8F && data[2] >= 0x80 && data[2] <= 0xBF && data[3] >= 0x80 && data[3] <= 0xBF)
|
|
{
|
|
return 4;
|
|
}
|
|
}
|
|
return 0; // invalid, incomplete, or must be diagnosed by the byte path
|
|
}
|
|
|
|
// Return the length of the longest prefix of [data, data+n) that consists of
|
|
// ASCII characters and complete well-formed UTF-8 sequences; n if all of it is
|
|
// valid UTF-8. Unlike scalar_string_bulk_run(), quotes, escapes, and control
|
|
// characters are ordinary characters here. ASCII is skipped 8 bytes at a time.
|
|
inline std::size_t valid_utf8_prefix(const unsigned char* data, std::size_t n) noexcept
|
|
{
|
|
constexpr std::uint64_t high = 0x8080808080808080ull;
|
|
std::size_t pos = 0;
|
|
while (pos < n)
|
|
{
|
|
if (pos + 8 <= n)
|
|
{
|
|
std::uint64_t word = 0;
|
|
std::memcpy(&word, data + pos, sizeof(word));
|
|
if ((word & high) == 0)
|
|
{
|
|
pos += 8;
|
|
continue;
|
|
}
|
|
}
|
|
|
|
if (data[pos] < 0x80u)
|
|
{
|
|
++pos;
|
|
continue;
|
|
}
|
|
|
|
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
|
if (seq == 0)
|
|
{
|
|
break; // ill-formed or truncated
|
|
}
|
|
pos += seq;
|
|
}
|
|
return pos;
|
|
}
|
|
|
|
// Scalar (C++11) computation of the bulk run length: the number of leading
|
|
// bytes in [data, data+n) that are ordinary ASCII or complete well-formed UTF-8
|
|
// sequences, stopping before the first byte that needs individual handling (the
|
|
// closing quote, an escape, a control character, or an ill-formed/truncated
|
|
// sequence). ASCII is skipped 8 bytes at a time.
|
|
inline std::size_t scalar_string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
|
{
|
|
std::size_t pos = 0;
|
|
while (pos < n)
|
|
{
|
|
pos += find_string_special(data + pos, n - pos);
|
|
if (pos >= n || data[pos] < 0x80u)
|
|
{
|
|
break; // end of buffer, or a quote/escape/control byte
|
|
}
|
|
// a run of multi-byte sequences (e.g. CJK text) is validated sequence
|
|
// by sequence without searching for the next special byte in between
|
|
do
|
|
{
|
|
const std::size_t seq = validate_one_utf8(data + pos, n - pos);
|
|
if (seq == 0)
|
|
{
|
|
return pos; // ill-formed or truncated: let the byte path diagnose it
|
|
}
|
|
pos += seq;
|
|
}
|
|
while (pos < n && data[pos] >= 0x80u);
|
|
}
|
|
return pos;
|
|
}
|
|
|
|
#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17)
|
|
// Index of the first quote/escape/control byte in [data, data+n) (non-ASCII
|
|
// bytes are *not* stops here - the whole run is handed to simdutf), or n.
|
|
inline std::size_t find_string_delimiter(const unsigned char* data, std::size_t n) noexcept
|
|
{
|
|
constexpr std::uint64_t ones = 0x0101010101010101ull;
|
|
constexpr std::uint64_t high = 0x8080808080808080ull;
|
|
std::size_t i = 0;
|
|
for (; i + 8 <= n; i += 8)
|
|
{
|
|
const std::uint64_t v = read_eight_bytes(data + i);
|
|
const std::uint64_t q = v ^ 0x2222222222222222ull;
|
|
const std::uint64_t b = v ^ 0x5C5C5C5C5C5C5C5Cull;
|
|
const std::uint64_t hit = ((q - ones) & ~q & high)
|
|
| ((b - ones) & ~b & high)
|
|
| ((v - 0x2020202020202020ull) & ~v & high);
|
|
if (hit != 0)
|
|
{
|
|
// the lowest flagged byte is the first delimiter (see find_string_special())
|
|
return i + (static_cast<std::size_t>(count_trailing_zeros(hit)) / 8);
|
|
}
|
|
}
|
|
for (; i < n; ++i)
|
|
{
|
|
const unsigned char c = data[i];
|
|
if (c == '\"' || c == '\\' || c < 0x20u)
|
|
{
|
|
return i;
|
|
}
|
|
}
|
|
return n;
|
|
}
|
|
#endif
|
|
|
|
// Backend-dispatched bulk run length. With JSON_USE_SIMDUTF the run up to the
|
|
// next delimiter is validated in one shot by simdutf; on the rare failure the
|
|
// scalar helper recomputes the exact valid prefix so the byte path still
|
|
// produces the precise diagnostic. Without it, the pure scalar path is used.
|
|
inline std::size_t string_bulk_run(const unsigned char* data, std::size_t n) noexcept
|
|
{
|
|
#if defined(JSON_USE_SIMDUTF) && defined(JSON_HAS_CPP_17)
|
|
const std::size_t run = find_string_delimiter(data, n);
|
|
if (run != 0 && simdutf::validate_utf8(reinterpret_cast<const char*>(data), run))
|
|
{
|
|
return run;
|
|
}
|
|
#endif
|
|
return scalar_string_bulk_run(data, n);
|
|
}
|
|
|
|
// Decode the 4 hex digits at [data, data+4) - the digits following a `\u`
|
|
// escape - into a codepoint 0x0000..0xFFFF via one table lookup per byte
|
|
// (after yyjson's read_hex_u16), or return -1 if any of the 4 bytes is not a
|
|
// hex digit ('0'..'9', 'A'..'F', 'a'..'f'). The caller must already have
|
|
// checked that 4 bytes are available; used by lexer::get_codepoint()'s
|
|
// contiguous fast path. On -1 it falls back to the byte-at-a-time loop, which
|
|
// stops at the first invalid digit, so the reported error and position are
|
|
// unaffected by this fast path.
|
|
inline int hex_codepoint(const unsigned char* data) noexcept
|
|
{
|
|
static const std::array<std::uint8_t, 256> hex_digit_table = // NOLINT(cppcoreguidelines-avoid-non-const-global-variables)
|
|
{
|
|
{
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 00..0F
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 10..1F
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 20..2F
|
|
0x00, 0x01, 0x02, 0x03, 0x04, 0x05, 0x06, 0x07, 0x08, 0x09, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 30..3F ('0'..'9')
|
|
0xFF, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 40..4F ('A'..'F')
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 50..5F
|
|
0xFF, 0x0A, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 60..6F ('a'..'f')
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 70..7F
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 80..8F
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // 90..9F
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // A0..AF
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // B0..BF
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // C0..CF
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // D0..DF
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, // E0..EF
|
|
0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF, 0xFF // F0..FF
|
|
}
|
|
};
|
|
|
|
const std::uint8_t d0 = hex_digit_table[data[0]];
|
|
const std::uint8_t d1 = hex_digit_table[data[1]];
|
|
const std::uint8_t d2 = hex_digit_table[data[2]];
|
|
const std::uint8_t d3 = hex_digit_table[data[3]];
|
|
// every valid digit is <= 0xF; the combined OR only exceeds it if at
|
|
// least one of the four bytes was not a hex digit (looked up as 0xFF)
|
|
if ((d0 | d1 | d2 | d3) > 0x0F)
|
|
{
|
|
return -1;
|
|
}
|
|
return (d0 << 12) | (d1 << 8) | (d2 << 4) | d3;
|
|
}
|
|
|
|
} // namespace detail
|
|
NLOHMANN_JSON_NAMESPACE_END
|