diff --git a/.github/labeler.yml b/.github/labeler.yml index 5708f84f5..7884e69fb 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -45,6 +45,23 @@ labels: - label: "aspect: binary formats" title: "(?i)(bson|cbor|msgpack|messagepack|ubjson|bjdata|bon8|binary format)" +- label: "aspect: json_view" + files: + - "include/nlohmann/json_view\\.hpp" + - "include/nlohmann/detail/view/.*" + - "single_include/nlohmann/json_view\\.hpp" + - "tests/src/unit-json_view.*" + - "tests/src/fuzzer-parse_json_view\\.cpp" + - "tools/amalgamate/config_json_view\\.json" + - "docs/mkdocs/docs/features/json_view\\.md" + - "docs/mkdocs/docs/api/basic_json_(document|view)/.*" + - "docs/mkdocs/docs/api/(ordered_)?json_(document|view)\\.md" + - "docs/mkdocs/docs/examples/(basic_json_(document|view)__|(ordered_)?json_(document|view)).*" + - "tests/benchmarks/src/benchmarks_view\\.cpp" + +- label: "aspect: json_view" + title: "(?i)(json_view|json_document|zero-copy)" + - label: "python" files: - "\\.py$" diff --git a/.github/workflows/check_amalgamation.yml b/.github/workflows/check_amalgamation.yml index 2f234d8be..cf4caaeb4 100644 --- a/.github/workflows/check_amalgamation.yml +++ b/.github/workflows/check_amalgamation.yml @@ -118,12 +118,15 @@ jobs: python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json.json -s . python3 $TOOL_DIR/amalgamate.py -c $TOOL_DIR/config_json_fwd.json -s . cp include/nlohmann/json_literals.hpp $INCLUDE_DIR/json_literals.hpp + # the configuration of json_view.hpp comes with the pull request until + # it is on develop; the tool itself is still develop's + python3 $TOOL_DIR/amalgamate.py -c $MAIN_DIR/tools/amalgamate/config_json_view.json -s . # the header list of the Bazel "json" target must match the files in include/ cmake -P cmake/scripts/gen_bazel_build_file.cmake ${{ github.workspace }}/venv/bin/astyle --project=tools/astyle/.astylerc --suffix=none --quiet \ - $INCLUDE_DIR/json.hpp $INCLUDE_DIR/json_fwd.hpp + $INCLUDE_DIR/json.hpp $INCLUDE_DIR/json_fwd.hpp $INCLUDE_DIR/json_view.hpp # fail loudly if a directory is renamed or removed: find would only warn # about the missing path and silently drop its files from the check diff --git a/BUILD.bazel b/BUILD.bazel index 8f0167de8..a7aa7228c 100644 --- a/BUILD.bazel +++ b/BUILD.bazel @@ -20,6 +20,7 @@ cc_library( hdrs = [ "include/nlohmann/adl_serializer.hpp", "include/nlohmann/byte_container_with_subtype.hpp", + "include/nlohmann/detail/abi_config.hpp", "include/nlohmann/detail/abi_macros.hpp", "include/nlohmann/detail/bit_ops.hpp", "include/nlohmann/detail/conversions/from_json.hpp", @@ -66,9 +67,21 @@ cc_library( "include/nlohmann/detail/string_escape.hpp", "include/nlohmann/detail/string_utils.hpp", "include/nlohmann/detail/value_t.hpp", + "include/nlohmann/detail/view/builder.hpp", + "include/nlohmann/detail/view/document_data.hpp", + "include/nlohmann/detail/view/errors.hpp", + "include/nlohmann/detail/view/input.hpp", + "include/nlohmann/detail/view/macro_scope.hpp", + "include/nlohmann/detail/view/macro_unscope.hpp", + "include/nlohmann/detail/view/materialize.hpp", + "include/nlohmann/detail/view/node.hpp", + "include/nlohmann/detail/view/number.hpp", + "include/nlohmann/detail/view/scan.hpp", + "include/nlohmann/detail/view/string_ref.hpp", "include/nlohmann/json.hpp", "include/nlohmann/json_fwd.hpp", "include/nlohmann/json_literals.hpp", + "include/nlohmann/json_view.hpp", "include/nlohmann/ordered_map.hpp", "include/nlohmann/thirdparty/hedley/hedley.hpp", "include/nlohmann/thirdparty/hedley/hedley_undef.hpp", @@ -81,6 +94,7 @@ cc_library( name = "singleheader-json", hdrs = [ "single_include/nlohmann/json.hpp", + "single_include/nlohmann/json_view.hpp", ], includes = ["single_include"], visibility = ["//visibility:public"], diff --git a/Makefile b/Makefile index 3a7fa5c41..5ada34764 100644 --- a/Makefile +++ b/Makefile @@ -23,6 +23,7 @@ AMALGAMATED_FILE=single_include/nlohmann/json.hpp AMALGAMATED_FWD_FILE=single_include/nlohmann/json_fwd.hpp # json_literals.hpp only includes , so it is copied verbatim AMALGAMATED_LITERALS_FILE=single_include/nlohmann/json_literals.hpp +AMALGAMATED_VIEW_FILE=single_include/nlohmann/json_view.hpp # the header with the argument-counting macros generated by tools/macro_builder MACRO_SCOPE_HPP=include/nlohmann/detail/macro_scope.hpp @@ -34,7 +35,7 @@ MACRO_SCOPE_HPP=include/nlohmann/detail/macro_scope.hpp # main target all: - @echo "amalgamate - amalgamate files single_include/nlohmann/json{,_fwd,_literals}.hpp from the include/nlohmann sources" + @echo "amalgamate - amalgamate files single_include/nlohmann/json{,_fwd,_literals,_view}.hpp from the include/nlohmann sources" @echo "BUILD.bazel - regenerate the Bazel BUILD file from the include/nlohmann sources" @echo "ChangeLog.md - generate ChangeLog file" @echo "check-amalgamation - check whether sources have been amalgamated and BUILD.bazel is up to date" @@ -87,10 +88,10 @@ install_astyle: # call the Artistic Style pretty printer on all source files pretty: install_astyle - $(ASTYLE) --project=tools/astyle/.astylerc $(SRCS) $(TESTS_SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) docs/mkdocs/docs/examples/*.cpp docs/mkdocs/docs/examples/*.hpp + $(ASTYLE) --project=tools/astyle/.astylerc $(SRCS) $(TESTS_SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) $(AMALGAMATED_VIEW_FILE) docs/mkdocs/docs/examples/*.cpp docs/mkdocs/docs/examples/*.hpp # create single header files and pretty print -amalgamate: $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) +amalgamate: $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) $(AMALGAMATED_VIEW_FILE) $(MAKE) pretty # call the amalgamation tool for json.hpp @@ -105,6 +106,9 @@ $(AMALGAMATED_FWD_FILE): $(SRCS) $(AMALGAMATED_LITERALS_FILE): include/nlohmann/json_literals.hpp cp include/nlohmann/json_literals.hpp $(AMALGAMATED_LITERALS_FILE) +# call the amalgamation tool for json_view.hpp (keeps including json.hpp) +$(AMALGAMATED_VIEW_FILE): $(SRCS) + tools/amalgamate/amalgamate.py -c tools/amalgamate/config_json_view.json -s . --verbose=yes # regenerate nlohmann_json.natvis from the ABI tags and version in include/nlohmann/detail/abi_macros.hpp natvis: python3 tools/generate_natvis/generate_natvis.py . @@ -129,13 +133,16 @@ check-amalgamation: @mv $(AMALGAMATED_FILE) $(AMALGAMATED_FILE)~ @mv $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_FWD_FILE)~ @mv $(AMALGAMATED_LITERALS_FILE) $(AMALGAMATED_LITERALS_FILE)~ + @mv $(AMALGAMATED_VIEW_FILE) $(AMALGAMATED_VIEW_FILE)~ @$(MAKE) amalgamate @diff $(AMALGAMATED_FILE) $(AMALGAMATED_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_FILE)~ $(AMALGAMATED_FILE) ; false) @diff $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_FWD_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) ; false) @diff $(AMALGAMATED_LITERALS_FILE) $(AMALGAMATED_LITERALS_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_LITERALS_FILE)~ $(AMALGAMATED_LITERALS_FILE) ; false) + @diff $(AMALGAMATED_VIEW_FILE) $(AMALGAMATED_VIEW_FILE)~ || (echo "===================================================================\n Amalgamation required! Please read the contribution guidelines\n in file .github/CONTRIBUTING.md.\n===================================================================" ; mv $(AMALGAMATED_VIEW_FILE)~ $(AMALGAMATED_VIEW_FILE) ; false) @mv $(AMALGAMATED_FILE)~ $(AMALGAMATED_FILE) @mv $(AMALGAMATED_FWD_FILE)~ $(AMALGAMATED_FWD_FILE) @mv $(AMALGAMATED_LITERALS_FILE)~ $(AMALGAMATED_LITERALS_FILE) + @mv $(AMALGAMATED_VIEW_FILE)~ $(AMALGAMATED_VIEW_FILE) @mv BUILD.bazel BUILD.bazel~ @$(MAKE) BUILD.bazel @diff BUILD.bazel BUILD.bazel~ || (echo "===================================================================\n BUILD.bazel is out of date! Please run 'make BUILD.bazel'.\n===================================================================" ; mv BUILD.bazel~ BUILD.bazel ; false) @@ -181,7 +188,7 @@ json.tar.xz: # We use `-X` to make the resulting ZIP file reproducible, see # . include.zip: BUILD.bazel - zip -9 --recurse-paths -X include.zip $(SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) BUILD.bazel MODULE.bazel meson.build LICENSE.MIT + zip -9 --recurse-paths -X include.zip $(SRCS) $(AMALGAMATED_FILE) $(AMALGAMATED_FWD_FILE) $(AMALGAMATED_LITERALS_FILE) $(AMALGAMATED_VIEW_FILE) BUILD.bazel MODULE.bazel meson.build LICENSE.MIT # Create the files for a release and add signatures and hashes. release: include.zip json.tar.xz @@ -191,11 +198,13 @@ release: include.zip json.tar.xz gpg --armor --detach-sig $(AMALGAMATED_FILE) gpg --armor --detach-sig $(AMALGAMATED_FWD_FILE) gpg --armor --detach-sig $(AMALGAMATED_LITERALS_FILE) + gpg --armor --detach-sig $(AMALGAMATED_VIEW_FILE) gpg --armor --detach-sig json.tar.xz cp $(AMALGAMATED_FILE) release_files cp $(AMALGAMATED_FWD_FILE) release_files cp $(AMALGAMATED_LITERALS_FILE) release_files - mv $(AMALGAMATED_FILE).asc $(AMALGAMATED_FWD_FILE).asc $(AMALGAMATED_LITERALS_FILE).asc json.tar.xz json.tar.xz.asc include.zip include.zip.asc release_files + cp $(AMALGAMATED_VIEW_FILE) release_files + mv $(AMALGAMATED_FILE).asc $(AMALGAMATED_FWD_FILE).asc $(AMALGAMATED_LITERALS_FILE).asc $(AMALGAMATED_VIEW_FILE).asc json.tar.xz json.tar.xz.asc include.zip include.zip.asc release_files cd release_files ; shasum -a 256 $$(find . -type f -not -name '*.asc' | sed 's|^\./||' | sort) > hashes.txt diff --git a/README.md b/README.md index 88cf88ae4..e80ede97d 100644 --- a/README.md +++ b/README.md @@ -1187,6 +1187,14 @@ binary.set_subtype(0x10); auto cbor = json::to_msgpack(j); // 0xD5 (fixext2), 0x10, 0xCA, 0xFE ``` +### Zero-copy views + +Header `` adds `json_document`/`json_view`, a read-only, non-owning way to look at a parsed +JSON text: parsing builds a flat index (16 bytes per value) instead of a tree, strings and numbers stay in the source +text, and `materialize()` builds a `json` value for a subtree only when you actually need one. See +[Zero-copy JSON views](https://json.nlohmann.me/features/json_view/) for the details, including which inputs are +borrowed and which are copied. + ## Customers The library is used in multiple projects, applications, operating systems, etc. The list below is not exhaustive, but the result of an internet search. If you know further customers of the library, please let me know, see [contact](#contact). @@ -1395,6 +1403,7 @@ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR I - The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). - The class contains parts of [Google Abseil](https://github.com/abseil/abseil-cpp) which is licensed under the [Apache 2.0 License](https://opensource.org/licenses/Apache-2.0). - The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors +- The view's parser (``) contains techniques and code adapted from [yyjson](https://github.com/ibireme/yyjson) by YaoYuan, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above): table-driven decoding of `\u` escapes and fixed-offset unrolled checks. REUSE Software diff --git a/cmake/ci.cmake b/cmake/ci.cmake index c4bced5ba..1a0a22386 100644 --- a/cmake/ci.cmake +++ b/cmake/ci.cmake @@ -410,10 +410,11 @@ list(FILTER INDENT_FILES EXCLUDE REGEX "/tests/thirdparty/|/tests/abi/include/nl set(include_dir ${PROJECT_SOURCE_DIR}/single_include/nlohmann) set(tool_dir ${PROJECT_SOURCE_DIR}/tools/amalgamate) add_custom_target(ci_test_amalgamation - COMMAND rm -fr ${include_dir}/json.hpp~ ${include_dir}/json_fwd.hpp~ ${include_dir}/json_literals.hpp~ + COMMAND rm -fr ${include_dir}/json.hpp~ ${include_dir}/json_fwd.hpp~ ${include_dir}/json_literals.hpp~ ${include_dir}/json_view.hpp~ COMMAND cp ${include_dir}/json.hpp ${include_dir}/json.hpp~ COMMAND cp ${include_dir}/json_fwd.hpp ${include_dir}/json_fwd.hpp~ COMMAND cp ${include_dir}/json_literals.hpp ${include_dir}/json_literals.hpp~ + COMMAND cp ${include_dir}/json_view.hpp ${include_dir}/json_view.hpp~ COMMAND cp ${PROJECT_SOURCE_DIR}/BUILD.bazel ${PROJECT_SOURCE_DIR}/BUILD.bazel~ COMMAND ${Python3_EXECUTABLE} -mvenv venv_astyle @@ -423,12 +424,14 @@ add_custom_target(ci_test_amalgamation COMMAND ${Python3_EXECUTABLE} ${tool_dir}/amalgamate.py -c ${tool_dir}/config_json.json -s . COMMAND ${Python3_EXECUTABLE} ${tool_dir}/amalgamate.py -c ${tool_dir}/config_json_fwd.json -s . COMMAND cp ${PROJECT_SOURCE_DIR}/include/nlohmann/json_literals.hpp ${include_dir}/json_literals.hpp - COMMAND venv_astyle/bin/astyle --project=tools/astyle/.astylerc --suffix=none ${include_dir}/json.hpp ${include_dir}/json_fwd.hpp + COMMAND ${Python3_EXECUTABLE} ${tool_dir}/amalgamate.py -c ${tool_dir}/config_json_view.json -s . + COMMAND venv_astyle/bin/astyle --project=tools/astyle/.astylerc --suffix=none ${include_dir}/json.hpp ${include_dir}/json_fwd.hpp ${include_dir}/json_view.hpp COMMAND ${CMAKE_COMMAND} -P ${PROJECT_SOURCE_DIR}/cmake/scripts/gen_bazel_build_file.cmake COMMAND diff ${include_dir}/json.hpp~ ${include_dir}/json.hpp COMMAND diff ${include_dir}/json_fwd.hpp~ ${include_dir}/json_fwd.hpp COMMAND diff ${include_dir}/json_literals.hpp~ ${include_dir}/json_literals.hpp + COMMAND diff ${include_dir}/json_view.hpp~ ${include_dir}/json_view.hpp COMMAND diff ${PROJECT_SOURCE_DIR}/BUILD.bazel~ ${PROJECT_SOURCE_DIR}/BUILD.bazel COMMAND venv_astyle/bin/astyle --project=tools/astyle/.astylerc --suffix=orig ${INDENT_FILES} diff --git a/cmake/scripts/gen_bazel_build_file.cmake b/cmake/scripts/gen_bazel_build_file.cmake index a781a2e3f..04c66823c 100644 --- a/cmake/scripts/gen_bazel_build_file.cmake +++ b/cmake/scripts/gen_bazel_build_file.cmake @@ -48,6 +48,7 @@ cc_library( name = "singleheader-json", hdrs = [ "single_include/nlohmann/json.hpp", + "single_include/nlohmann/json_view.hpp", ], includes = ["single_include"], visibility = ["//visibility:public"], diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 8e956413c..6f46a2f15 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -132,7 +132,43 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::to_ubjson', 'Func INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::value', 'Method', 'api/basic_json/value/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::value_t', 'Enum', 'api/basic_json/value_t/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json::~basic_json', 'Method', 'api/basic_json/~basic_json/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document', 'Class', 'api/basic_json_document/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::basic_json_document', 'Constructor', 'api/basic_json_document/basic_json_document/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::accept', 'Function', 'api/basic_json_document/accept/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::is_discarded', 'Method', 'api/basic_json_document/is_discarded/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::memory_usage', 'Method', 'api/basic_json_document/memory_usage/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::node_count', 'Method', 'api/basic_json_document/node_count/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::owns_source', 'Method', 'api/basic_json_document/owns_source/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::parse', 'Function', 'api/basic_json_document/parse/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::parse_copy', 'Function', 'api/basic_json_document/parse_copy/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::read', 'Method', 'api/basic_json_document/read/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::root', 'Method', 'api/basic_json_document/root/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::shrink_to_fit', 'Method', 'api/basic_json_document/shrink_to_fit/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::source', 'Method', 'api/basic_json_document/source/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view', 'Class', 'api/basic_json_view/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::basic_json_view', 'Constructor', 'api/basic_json_view/basic_json_view/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::empty', 'Method', 'api/basic_json_view/empty/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_array', 'Method', 'api/basic_json_view/is_array/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_binary', 'Method', 'api/basic_json_view/is_binary/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_boolean', 'Method', 'api/basic_json_view/is_boolean/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_discarded', 'Method', 'api/basic_json_view/is_discarded/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_null', 'Method', 'api/basic_json_view/is_null/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_number', 'Method', 'api/basic_json_view/is_number/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_number_float', 'Method', 'api/basic_json_view/is_number_float/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_number_integer', 'Method', 'api/basic_json_view/is_number_integer/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_number_unsigned', 'Method', 'api/basic_json_view/is_number_unsigned/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_object', 'Method', 'api/basic_json_view/is_object/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_primitive', 'Method', 'api/basic_json_view/is_primitive/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_string', 'Method', 'api/basic_json_view/is_string/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::is_structured', 'Method', 'api/basic_json_view/is_structured/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::materialize', 'Method', 'api/basic_json_view/materialize/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::operator bool', 'Method', 'api/basic_json_view/operator_bool/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::size', 'Method', 'api/basic_json_view/size/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::source_offset', 'Method', 'api/basic_json_view/source_offset/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_view::type', 'Method', 'api/basic_json_view/type/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('json', 'Class', 'api/json/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('json_document', 'Class', 'api/json_document/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('json_view', 'Class', 'api/json_view/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer', 'Class', 'api/json_pointer/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::back', 'Method', 'api/json_pointer/back/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('json_pointer::empty', 'Method', 'api/json_pointer/empty/index.html'); @@ -170,6 +206,8 @@ INSERT INTO searchIndex(name, type, path) VALUES ('operator""_json_pointer', 'Li INSERT INTO searchIndex(name, type, path) VALUES ('operator<<', 'Operator', 'api/operator_ltlt/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('operator>>', 'Operator', 'api/operator_gtgt/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('ordered_json', 'Class', 'api/ordered_json/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('ordered_json_document', 'Class', 'api/ordered_json_document/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('ordered_json_view', 'Class', 'api/ordered_json_view/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('ordered_map', 'Class', 'api/ordered_map/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('std::formatter', 'Class', 'api/basic_json/std_formatter/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('std::hash', 'Class', 'api/basic_json/std_hash/index.html'); @@ -200,6 +238,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('Iterators', 'Guide', 'feature INSERT INTO searchIndex(name, type, path) VALUES ('JSON Merge Patch', 'Guide', 'features/merge_patch/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON Patch and Diff', 'Guide', 'features/json_patch/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('JSON Pointer', 'Guide', 'features/json_pointer/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('Zero-copy JSON views', 'Guide', 'features/json_view/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('nlohmann Namespace', 'Guide', 'features/namespace/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('Types', 'Guide', 'features/types/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('Types: Number Handling', 'Guide', 'features/types/number_handling/index.html'); diff --git a/docs/mkdocs/docs/api/basic_json/empty.md b/docs/mkdocs/docs/api/basic_json/empty.md index 1d393d76e..1f096c956 100644 --- a/docs/mkdocs/docs/api/basic_json/empty.md +++ b/docs/mkdocs/docs/api/basic_json/empty.md @@ -64,6 +64,7 @@ itself is empty which is `#!cpp false` in the case of a string. - [size](size.md) returns the number of elements - [clear](clear.md) clears the content and resets the value to the default value +- [basic_json_view::empty](../basic_json_view/empty.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_array.md b/docs/mkdocs/docs/api/basic_json/is_array.md index af68bdc0d..9f2bfc6fc 100644 --- a/docs/mkdocs/docs/api/basic_json/is_array.md +++ b/docs/mkdocs/docs/api/basic_json/is_array.md @@ -40,6 +40,7 @@ Constant. - [is_structured](is_structured.md) checks whether the JSON value is structured (array or object) - [type](type.md) returns the type of the JSON value - [array_t](array_t.md) the type used to store JSON arrays +- [basic_json_view::is_array](../basic_json_view/is_array.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_binary.md b/docs/mkdocs/docs/api/basic_json/is_binary.md index ecd853fcd..87bb8b66c 100644 --- a/docs/mkdocs/docs/api/basic_json/is_binary.md +++ b/docs/mkdocs/docs/api/basic_json/is_binary.md @@ -39,6 +39,7 @@ Constant. - [is_primitive](is_primitive.md) checks whether the JSON value is primitive - [binary_t](binary_t.md) the type used to store binary values - [get_binary](get_binary.md) returns a reference to the stored binary value +- [basic_json_view::is_binary](../basic_json_view/is_binary.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_boolean.md b/docs/mkdocs/docs/api/basic_json/is_boolean.md index 69e85cbfe..c7585812c 100644 --- a/docs/mkdocs/docs/api/basic_json/is_boolean.md +++ b/docs/mkdocs/docs/api/basic_json/is_boolean.md @@ -38,6 +38,7 @@ Constant. - [boolean_t](boolean_t.md) the type used to store JSON booleans - [is_primitive](is_primitive.md) checks whether the JSON value is primitive +- [basic_json_view::is_boolean](../basic_json_view/is_boolean.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_discarded.md b/docs/mkdocs/docs/api/basic_json/is_discarded.md index 9ab0ab38a..3d434f577 100644 --- a/docs/mkdocs/docs/api/basic_json/is_discarded.md +++ b/docs/mkdocs/docs/api/basic_json/is_discarded.md @@ -85,6 +85,11 @@ with `allow_exceptions` set to `#!cpp false`: a parse error then yields a discar --8<-- "examples/is_discarded__parse.output" ``` +## See also + +- [basic_json_view::is_discarded](../basic_json_view/is_discarded.md) - the corresponding check on a zero-copy view, + which is `#!cpp true` if the view refers to no value + ## Version history - Added in version 1.0.0. diff --git a/docs/mkdocs/docs/api/basic_json/is_null.md b/docs/mkdocs/docs/api/basic_json/is_null.md index e90a6c7ea..3823b96dd 100644 --- a/docs/mkdocs/docs/api/basic_json/is_null.md +++ b/docs/mkdocs/docs/api/basic_json/is_null.md @@ -40,6 +40,7 @@ Constant. - [is_object](is_object.md) checks whether the JSON value is an object - [type](type.md) returns the type of the JSON value - [value_t](value_t.md) the enumeration of JSON types +- [basic_json_view::is_null](../basic_json_view/is_null.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_number.md b/docs/mkdocs/docs/api/basic_json/is_number.md index afb30bb0c..5db28d3e0 100644 --- a/docs/mkdocs/docs/api/basic_json/is_number.md +++ b/docs/mkdocs/docs/api/basic_json/is_number.md @@ -49,6 +49,7 @@ constexpr bool is_number() const noexcept - [is_number_integer()](is_number_integer.md) check if the value is an integer or unsigned integer number - [is_number_unsigned()](is_number_unsigned.md) check if the value is an unsigned integer number - [is_number_float()](is_number_float.md) check if the value is a floating-point number +- [basic_json_view::is_number](../basic_json_view/is_number.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_number_float.md b/docs/mkdocs/docs/api/basic_json/is_number_float.md index ad18858e3..c94825b41 100644 --- a/docs/mkdocs/docs/api/basic_json/is_number_float.md +++ b/docs/mkdocs/docs/api/basic_json/is_number_float.md @@ -40,6 +40,7 @@ Constant. - [is_number()](is_number.md) check if the value is a number - [is_number_integer()](is_number_integer.md) check if the value is an integer or unsigned integer number - [is_number_unsigned()](is_number_unsigned.md) check if the value is an unsigned integer number +- [basic_json_view::is_number_float](../basic_json_view/is_number_float.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_number_integer.md b/docs/mkdocs/docs/api/basic_json/is_number_integer.md index b8f971bd6..71e2aba20 100644 --- a/docs/mkdocs/docs/api/basic_json/is_number_integer.md +++ b/docs/mkdocs/docs/api/basic_json/is_number_integer.md @@ -40,6 +40,7 @@ Constant. - [is_number()](is_number.md) check if the value is a number - [is_number_unsigned()](is_number_unsigned.md) check if the value is an unsigned integer number - [is_number_float()](is_number_float.md) check if the value is a floating-point number +- [basic_json_view::is_number_integer](../basic_json_view/is_number_integer.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_number_unsigned.md b/docs/mkdocs/docs/api/basic_json/is_number_unsigned.md index 50083164e..3e0720ef1 100644 --- a/docs/mkdocs/docs/api/basic_json/is_number_unsigned.md +++ b/docs/mkdocs/docs/api/basic_json/is_number_unsigned.md @@ -40,6 +40,7 @@ Constant. - [is_number()](is_number.md) check if the value is a number - [is_number_integer()](is_number_integer.md) check if the value is an integer or unsigned integer number - [is_number_float()](is_number_float.md) check if the value is a floating-point number +- [basic_json_view::is_number_unsigned](../basic_json_view/is_number_unsigned.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_object.md b/docs/mkdocs/docs/api/basic_json/is_object.md index 17776a8bc..2105a28a7 100644 --- a/docs/mkdocs/docs/api/basic_json/is_object.md +++ b/docs/mkdocs/docs/api/basic_json/is_object.md @@ -40,6 +40,7 @@ Constant. - [is_structured](is_structured.md) checks whether the JSON value is structured (array or object) - [type](type.md) returns the type of the JSON value - [object_t](object_t.md) the type used to store JSON objects +- [basic_json_view::is_object](../basic_json_view/is_object.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_primitive.md b/docs/mkdocs/docs/api/basic_json/is_primitive.md index 076c2bfef..70836ecdd 100644 --- a/docs/mkdocs/docs/api/basic_json/is_primitive.md +++ b/docs/mkdocs/docs/api/basic_json/is_primitive.md @@ -62,6 +62,7 @@ This library extends primitive types to binary types, because binary types are r - [is_boolean()](is_boolean.md) returns whether the JSON value is a boolean - [is_number()](is_number.md) returns whether the JSON value is a number - [is_binary()](is_binary.md) returns whether the JSON value is a binary array +- [basic_json_view::is_primitive](../basic_json_view/is_primitive.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_string.md b/docs/mkdocs/docs/api/basic_json/is_string.md index d91eaf0a8..a1d19c14d 100644 --- a/docs/mkdocs/docs/api/basic_json/is_string.md +++ b/docs/mkdocs/docs/api/basic_json/is_string.md @@ -39,6 +39,7 @@ Constant. - [is_primitive](is_primitive.md) checks whether the JSON value is primitive - [type](type.md) returns the type of the JSON value - [string_t](string_t.md) the type used to store JSON strings +- [basic_json_view::is_string](../basic_json_view/is_string.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/is_structured.md b/docs/mkdocs/docs/api/basic_json/is_structured.md index f2d9e79ff..a67f2778b 100644 --- a/docs/mkdocs/docs/api/basic_json/is_structured.md +++ b/docs/mkdocs/docs/api/basic_json/is_structured.md @@ -57,6 +57,7 @@ Note that though strings are containers in C++, they are treated as primitive va - [is_primitive()](is_primitive.md) returns whether JSON value is primitive - [is_array()](is_array.md) returns whether the value is an array - [is_object()](is_object.md) returns whether the value is an object +- [basic_json_view::is_structured](../basic_json_view/is_structured.md) - the same check on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/size.md b/docs/mkdocs/docs/api/basic_json/size.md index c3d9156ba..8fe553bd3 100644 --- a/docs/mkdocs/docs/api/basic_json/size.md +++ b/docs/mkdocs/docs/api/basic_json/size.md @@ -55,6 +55,7 @@ JSON value which is `1` in the case of a string. - [empty](empty.md) checks whether the JSON value has no elements - [max_size](max_size.md) returns the maximum possible number of elements +- [basic_json_view::size](../basic_json_view/size.md) - the same function on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json/type.md b/docs/mkdocs/docs/api/basic_json/type.md index 381e489d2..6ed2aaeda 100644 --- a/docs/mkdocs/docs/api/basic_json/type.md +++ b/docs/mkdocs/docs/api/basic_json/type.md @@ -52,6 +52,7 @@ Constant. - [operator value_t](operator_value_t.md) implicit conversion operator equivalent to this named member function - [type_name](type_name.md) returns the type as a string, for use in error messages - [value_t](value_t.md) the enumeration of JSON types +- [basic_json_view::type](../basic_json_view/type.md) - the same function on a zero-copy view ## Version history diff --git a/docs/mkdocs/docs/api/basic_json_document/accept.md b/docs/mkdocs/docs/api/basic_json_document/accept.md new file mode 100644 index 000000000..5aa1fc93d --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/accept.md @@ -0,0 +1,67 @@ +# nlohmann::basic_json_document::accept + +```cpp +template +static bool accept(InputType&& input, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false); +``` + +Checks whether the input is valid JSON, accepting and rejecting exactly what +[`BasicJsonType::accept()`](../basic_json/accept.md) does, with the same options. Unlike [`parse()`](parse.md), this +function never throws an exception for invalid input, and the returned `#!cpp bool` is the only result -- no document +is returned. + +## Template parameters + +`InputType` +: A compatible input; see [`parse`](parse.md#template-parameters). + +## Parameters + +`input` (in) +: Input to check. + +`ignore_comments` (in) +: whether comments should be ignored and treated like whitespace (`#!cpp true`) or yield a parse error + (`#!cpp false`); (optional, `#!cpp false` by default) + +`ignore_trailing_commas` (in) +: whether trailing commas in arrays or objects should be ignored and treated like whitespace (`#!cpp true`) or + yield a parse error (`#!cpp false`); (optional, `#!cpp false` by default) + +## Return value + +Whether the input is valid JSON. + +## Exception safety + +Strong guarantee: this function itself never throws for an invalid input; it can only throw what allocating the +input's own copy (for inputs that are always read into a buffer) throws. + +## Complexity + +Linear in the length of the input. + +## Examples + +??? example + + ```cpp + --8<-- "examples/basic_json_document__accept.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__accept.output" + ``` + +## See also + +- [parse](parse.md) - deserialize from a compatible input +- [`BasicJsonType::accept`](../basic_json/accept.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/basic_json_document.md b/docs/mkdocs/docs/api/basic_json_document/basic_json_document.md new file mode 100644 index 000000000..b877e8895 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/basic_json_document.md @@ -0,0 +1,58 @@ +# nlohmann::basic_json_document::basic_json_document + +```cpp +// (1) +basic_json_document() = default; + +// (2) +basic_json_document(basic_json_document&& other) noexcept = default; + +// (3) +basic_json_document(const basic_json_document&) = delete; +``` + +1. Creates an empty (discarded) document: [`root()`](root.md) returns a discarded view, and + [`is_discarded()`](is_discarded.md) is `#!cpp true`. +2. Move constructor. Takes over `other`'s index and, if owned, its text; `other` is left as an empty document. Views + taken from `other` before the move remain valid, because the index is heap-allocated independently of the + `basic_json_document` object. +3. `basic_json_document` is move-only. Copying is disabled because it would either duplicate a potentially large index + and text, or leave two documents claiming to borrow the same buffer. + +## Parameters + +`other` (in) +: another document to move the index and text from + +## Exception safety + +No-throw guarantee: the default and move constructors never throw exceptions. + +## Complexity + +Constant, for the default and move constructors. + +## Examples + +??? example + + The example below shows the default constructor and demonstrates that `basic_json_document` is move-only. + + ```cpp + --8<-- "examples/basic_json_document__basic_json_document.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__basic_json_document.output" + ``` + +## See also + +- [parse](parse.md) - deserialize from a compatible input +- [is_discarded](is_discarded.md) - return whether the last parse failed + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/index.md b/docs/mkdocs/docs/api/basic_json_document/index.md new file mode 100644 index 000000000..89564369a --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/index.md @@ -0,0 +1,56 @@ +# nlohmann::basic_json_document + +Defined in header `` + +```cpp +template +class basic_json_document; +``` + +A parsed JSON text, held as a flat index of its values +([16 bytes per value](../../home/architecture.md#node-index-of-json-views)) instead of a tree of `BasicJsonType` values. +Strings and numbers stay in the source text; only strings that contain escapes are decoded, into one buffer owned by +the document. [`basic_json_view`](../basic_json_view/index.md) is a read-only handle to one value of a +`basic_json_document`; [`materialize()`](../basic_json_view/materialize.md) turns a subtree back into the +`BasicJsonType` value that [`BasicJsonType::parse()`](../basic_json/parse.md) would have produced for it. + +A document may **borrow** the text it was parsed from (the caller's buffer must then outlive the document) or **own** +it (a copy, or an rvalue `#!cpp std::string` that was moved in); see [`owns_source`](owns_source.md). `basic_json_document` +is move-only: copying a document would either duplicate a potentially large index and text, or leave two documents +claiming to borrow the same buffer, so it is disabled. + +## Template parameters + +`BasicJsonType` +: a specialization of [`basic_json`](../basic_json/index.md), for instance [`json`](../json.md) or + [`ordered_json`](../ordered_json.md). Only 64-bit `number_integer_t`/`number_unsigned_t` types are supported; this + is checked with a `static_assert`. + +## Specializations + +- [**json_document**](../json_document.md) - documents of the default specialization [`json`](../json.md) +- [**ordered_json_document**](../ordered_json_document.md) - documents of [`ordered_json`](../ordered_json.md) + +## Member types + +- **view_type** - the type of view returned by [`root()`](root.md) (`#!cpp basic_json_view`) +- **value_t** - the JSON type enumeration, see [`basic_json::value_t`](../basic_json/value_t.md) + +## Member functions + +- [(constructor)](basic_json_document.md) +- [**parse**](parse.md) (_static_) - deserialize from a compatible input, borrowing or owning it as appropriate +- [**parse_copy**](parse_copy.md) (_static_) - deserialize a copy of a compatible input +- [**accept**](accept.md) (_static_) - check whether the input is valid JSON +- [**read**](read.md) - (re-)parse into this document, reusing its memory +- [**root**](root.md) - the view of the root value +- [**is_discarded**](is_discarded.md) - return whether the last parse failed +- [**source**](source.md) - the parsed text +- [**owns_source**](owns_source.md) - return whether the document holds its own copy of the text +- [**node_count**](node_count.md) - the number of index entries (values plus object keys) +- [**memory_usage**](memory_usage.md) - the number of bytes held by the document +- [**shrink_to_fit**](shrink_to_fit.md) - release unused index capacity + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/is_discarded.md b/docs/mkdocs/docs/api/basic_json_document/is_discarded.md new file mode 100644 index 000000000..3c062dddd --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/is_discarded.md @@ -0,0 +1,49 @@ +# nlohmann::basic_json_document::is_discarded + +```cpp +bool is_discarded() const noexcept; +``` + +Returns whether the document holds no value, either because it was default-constructed or because the last call to +[`parse()`](parse.md), [`parse_copy()`](parse_copy.md), or [`read()`](read.md) failed with `allow_exceptions` set to +`#!cpp false`. + +## Return value + +`#!cpp true` if the document is discarded, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +When the document is discarded, [`root()`](root.md) returns a discarded view (its +[`is_discarded()`](../basic_json_view/is_discarded.md) is also `#!cpp true`). + +## Examples + +??? example + + ```cpp + --8<-- "examples/basic_json_document__is_discarded.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__is_discarded.output" + ``` + +## See also + +- [parse](parse.md) - deserialize from a compatible input +- [is_discarded (basic_json_view)](../basic_json_view/is_discarded.md) - return whether a view is invalid + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/memory_usage.md b/docs/mkdocs/docs/api/basic_json_document/memory_usage.md new file mode 100644 index 000000000..5ad25911b --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/memory_usage.md @@ -0,0 +1,50 @@ +# nlohmann::basic_json_document::memory_usage + +```cpp +std::size_t memory_usage() const noexcept; +``` + +Returns the number of bytes held by the document: the node index, the decoded-string buffer (for strings that +contain escapes), and, for an owned document, its copy of the source text. + +## Return value + +The number of bytes the document holds, `0` for a [discarded](is_discarded.md) document. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +The exact value depends on the platform, the allocator, and the library's own layout, and may change between +versions; do not rely on it being a specific number, and do not compare it across different builds or platforms. +Compare it for the same document over time, or between documents built with the same binary, instead -- for instance +to observe the effect of [`shrink_to_fit()`](shrink_to_fit.md). + +## Examples + +??? example + + ```cpp + --8<-- "examples/basic_json_document__memory_usage.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__memory_usage.output" + ``` + +## See also + +- [node_count](node_count.md) - the number of index entries +- [shrink_to_fit](shrink_to_fit.md) - release unused index capacity + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/node_count.md b/docs/mkdocs/docs/api/basic_json_document/node_count.md new file mode 100644 index 000000000..61923f8cd --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/node_count.md @@ -0,0 +1,49 @@ +# nlohmann::basic_json_document::node_count + +```cpp +std::size_t node_count() const noexcept; +``` + +Returns the number of entries in the document's flat index. + +## Return value + +The number of index entries: one per value (of any type, at any nesting depth) plus one per object key. `0` for a +[discarded](is_discarded.md) document. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +Each index entry is [16 bytes](../../home/architecture.md#node-index-of-json-views), so `#!cpp node_count() * 16` is +the size of the index itself (part, but not all, of [`memory_usage()`](memory_usage.md), which also counts decoded +strings and, for an owned document, the text). + +## Examples + +??? example + + ```cpp + --8<-- "examples/basic_json_document__node_count.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__node_count.output" + ``` + +## See also + +- [memory_usage](memory_usage.md) - the number of bytes held by the document +- [shrink_to_fit](shrink_to_fit.md) - release unused index capacity + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/owns_source.md b/docs/mkdocs/docs/api/basic_json_document/owns_source.md new file mode 100644 index 000000000..ab1fdb45e --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/owns_source.md @@ -0,0 +1,49 @@ +# nlohmann::basic_json_document::owns_source + +```cpp +bool owns_source() const noexcept; +``` + +Returns whether the document holds its own copy of the parsed text, as opposed to borrowing the caller's buffer. + +## Return value + +`#!cpp true` if the document owns the text returned by [`source()`](source.md), `#!cpp false` if it borrows it (or if +the document is [discarded](is_discarded.md)). + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +See the ownership table on [`parse`](parse.md#notes) for which inputs are borrowed and which are owned. A borrowed +document (`#!cpp owns_source() == false`) is only valid while the buffer it was parsed from is still alive. + +## Examples + +??? example + + ```cpp + --8<-- "examples/basic_json_document__owns_source.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__owns_source.output" + ``` + +## See also + +- [parse](parse.md) - deserialize from a compatible input +- [parse_copy](parse_copy.md) - deserialize a copy of a compatible input, always owned +- [source](source.md) - the parsed text + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/parse.md b/docs/mkdocs/docs/api/basic_json_document/parse.md new file mode 100644 index 000000000..a3c44f4e0 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/parse.md @@ -0,0 +1,139 @@ +# nlohmann::basic_json_document::parse + +```cpp +// (1) +template +static basic_json_document parse(InputType&& input, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false); + +// (2) +template +static basic_json_document parse(IteratorType first, IteratorType last, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false); +``` + +1. Deserialize from a compatible input, borrowing or owning it depending on its value category and type (see Notes). +2. Deserialize from a pair of input iterators. + +Both overloads accept exactly what [`BasicJsonType::parse()`](../basic_json/parse.md) accepts, with the same +`ignore_comments`/`ignore_trailing_commas` options, but build a [`basic_json_document`](index.md) (a flat index into +the input) instead of a tree of `BasicJsonType` values. + +## Template parameters + +`InputType` +: A compatible input, for instance: + + - a `#!cpp std::string`, `#!cpp std::string_view`, or a C-style array of characters + - a pointer to a null-terminated string of single byte characters + - a container for which `#!cpp obj.data()` and `#!cpp obj.size()` give contiguous single-byte access, e.g. + `#!cpp std::vector` or `#!cpp std::vector` + - an `#!cpp std::istream` object, or anything else [`BasicJsonType::parse()`](../basic_json/parse.md) accepts + +`IteratorType` +: a compatible iterator type, for instance a pair of pointers such as `ptr` and `ptr + len`, or a pair of + `#!cpp std::string::iterator` + +## Parameters + +`input` (in) +: Input to parse from. + +`allow_exceptions` (in) +: whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) + +`ignore_comments` (in) +: whether comments should be ignored and treated like whitespace (`#!cpp true`) or yield a parse error + (`#!cpp false`); (optional, `#!cpp false` by default) + +`ignore_trailing_commas` (in) +: whether trailing commas in arrays or objects should be ignored and treated like whitespace (`#!cpp true`) or + yield a parse error (`#!cpp false`); (optional, `#!cpp false` by default) + +`first` (in) +: iterator to the start of a character range + +`last` (in) +: iterator to the end of a character range + +## Return value + +The parsed document. If `allow_exceptions` is `#!cpp false` and the input is not valid JSON, the returned document is +discarded; see [`is_discarded`](is_discarded.md). + +## Exceptions + +Throws the same exception [`BasicJsonType::parse()`](../basic_json/parse.md) throws for the same input and options -- +the same exception id, message, and position -- because on a failing input the library's own parser is run on the +same bytes to produce the diagnostic. Additionally throws +[`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the input is 4 GiB or larger, a size +[`BasicJsonType::parse()`](../basic_json/parse.md) does not reject. + +## Complexity + +Linear in the length of the input. + +## Notes + +**Ownership.** Whether the document borrows `input` or owns a copy of it depends on its value category and type: + +| `input` | ownership | +|--------------------------------------------------------------------------------------|--------------------------------------------------------------| +| lvalue byte container (`std::string`, `std::vector`, ...), `std::string_view`, C string, character array | **borrowed** -- `input` must outlive the document | +| rvalue `#!cpp std::string` | **owned**, moved in without a copy | +| rvalue byte container other than `#!cpp std::string` | **owned**, copied | +| stream, wide string, or anything else read through the general input adapter | **owned**, read into a buffer (a stream is read to its end) | + +For overload (2), a pair of pointers to single-byte integers (e.g. `#!cpp const char*`, `#!cpp std::uint8_t*`) is +borrowed. From C++20 on, so is any other contiguous iterator over single bytes, such as +`#!cpp std::vector::iterator` or `#!cpp std::string::const_iterator`. Before C++20 these iterators cannot be +told apart from other class-type iterators, so their range is read into an owned buffer, as is any non-contiguous +range (e.g. of a `#!cpp std::list`). + +See [`owns_source`](owns_source.md) to check which happened after a call, and the +[feature page](../../features/json_view.md) for the reasoning. + +**Numbers.** As for [`BasicJsonType::parse()`](../basic_json/parse.md), an integer literal too large for the 64-bit +integer type becomes a floating-point value. + +## Examples + +??? example "Example: (1) borrowed vs. owned input, and errors identical to `BasicJsonType::parse()`" + + ```cpp + --8<-- "examples/basic_json_document__parse.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__parse.output" + ``` + +??? example "Example: (2) parse an iterator range (no NUL terminator required)" + + ```cpp + --8<-- "examples/basic_json_document__parse_iterator_pair.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__parse_iterator_pair.output" + ``` + +## See also + +- [parse_copy](parse_copy.md) - deserialize a copy of a compatible input +- [accept](accept.md) - check whether the input is valid JSON +- [read](read.md) - (re-)parse into this document, reusing its memory +- [owns_source](owns_source.md) - return whether the document holds its own copy of the text +- [`BasicJsonType::parse`](../basic_json/parse.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/parse_copy.md b/docs/mkdocs/docs/api/basic_json_document/parse_copy.md new file mode 100644 index 000000000..ed4714d22 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/parse_copy.md @@ -0,0 +1,77 @@ +# nlohmann::basic_json_document::parse_copy + +```cpp +template +static basic_json_document parse_copy(InputType&& input, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false); +``` + +Deserialize from a compatible input, always taking the document's own copy of it, regardless of the value category or +type of `input`. Unlike [`parse()`](parse.md), the returned document never depends on `input` staying alive. + +## Template parameters + +`InputType` +: A compatible input; see [`parse`](parse.md#template-parameters). + +## Parameters + +`input` (in) +: Input to parse from. + +`allow_exceptions` (in) +: whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) + +`ignore_comments` (in) +: whether comments should be ignored and treated like whitespace (`#!cpp true`) or yield a parse error + (`#!cpp false`); (optional, `#!cpp false` by default) + +`ignore_trailing_commas` (in) +: whether trailing commas in arrays or objects should be ignored and treated like whitespace (`#!cpp true`) or + yield a parse error (`#!cpp false`); (optional, `#!cpp false` by default) + +## Return value + +The parsed document, with [`owns_source()`](owns_source.md) `#!cpp true`. If `allow_exceptions` is `#!cpp false` and +the input is not valid JSON, the returned document is discarded; see [`is_discarded`](is_discarded.md). + +## Exceptions + +Same as [`parse`](parse.md#exceptions). + +## Complexity + +Linear in the length of the input. + +## Notes + +`parse_copy()` accepts and rejects exactly what [`parse()`](parse.md) does, and classifies numbers the same way; it +only differs in that the input is always copied rather than sometimes borrowed. Prefer [`parse()`](parse.md) when the +input's lifetime already covers the document's, since it avoids the copy for borrowed inputs. + +## Examples + +??? example + + The example below returns a document from a function whose local buffer would otherwise not outlive it. + + ```cpp + --8<-- "examples/basic_json_document__parse_copy.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__parse_copy.output" + ``` + +## See also + +- [parse](parse.md) - deserialize from a compatible input, borrowing it where possible +- [owns_source](owns_source.md) - return whether the document holds its own copy of the text + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/read.md b/docs/mkdocs/docs/api/basic_json_document/read.md new file mode 100644 index 000000000..2267e46bd --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/read.md @@ -0,0 +1,76 @@ +# nlohmann::basic_json_document::read + +```cpp +template +void read(InputType&& input, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false); +``` + +(Re-)parses `input` into `#!cpp *this`, discarding the document's previous value and reusing its memory (the node +index, the decoded-string buffer, and, if applicable, the owned copy of the text) rather than allocating a fresh +document. [`parse()`](parse.md) is implemented in terms of this function, applied to a default-constructed document. + +## Template parameters + +`InputType` +: A compatible input; see [`parse`](parse.md#template-parameters). + +## Parameters + +`input` (in) +: Input to parse from. + +`allow_exceptions` (in) +: whether to throw exceptions in case of a parse error (optional, `#!cpp true` by default) + +`ignore_comments` (in) +: whether comments should be ignored and treated like whitespace (`#!cpp true`) or yield a parse error + (`#!cpp false`); (optional, `#!cpp false` by default) + +`ignore_trailing_commas` (in) +: whether trailing commas in arrays or objects should be ignored and treated like whitespace (`#!cpp true`) or + yield a parse error (`#!cpp false`); (optional, `#!cpp false` by default) + +## Exceptions + +Same as [`parse`](parse.md#exceptions). + +## Complexity + +Linear in the length of the input. + +## Notes + +Every view taken from `#!cpp *this` before the call -- including the previous [`root()`](root.md) -- is invalidated, +whether or not the new parse succeeds; take fresh views from [`root()`](root.md) afterward. + +`input` is borrowed or owned by the same rules as [`parse()`](parse.md#notes); a document can borrow on one call and +own on the next, since ownership is decided freshly each time. + +## Examples + +??? example + + The example below parses a sequence of messages into the same document, reusing its memory instead of allocating + a new document for each one. + + ```cpp + --8<-- "examples/basic_json_document__read.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__read.output" + ``` + +## See also + +- [parse](parse.md) - deserialize from a compatible input +- [root](root.md) - the view of the root value + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/root.md b/docs/mkdocs/docs/api/basic_json_document/root.md new file mode 100644 index 000000000..85682dbcd --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/root.md @@ -0,0 +1,50 @@ +# nlohmann::basic_json_document::root + +```cpp +view_type root() const noexcept; +``` + +Returns a view of the root value of the document. + +## Return value + +A [`view_type`](index.md#member-types) (i.e. `#!cpp basic_json_view`) for the root value, or a +discarded view if the document is [discarded](is_discarded.md). + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +`root()` is a cheap handle into the document's index, not a copy of anything; call it as often as needed. The +returned view is valid under the same conditions as any other view of the document -- see +[Object inspection](../basic_json_view/index.md) -- in particular, it is invalidated by the next +[`read()`](read.md) or [`shrink_to_fit()`](shrink_to_fit.md) on this document. + +## Examples + +??? example + + ```cpp + --8<-- "examples/basic_json_document__root.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__root.output" + ``` + +## See also + +- [is_discarded](is_discarded.md) - return whether the last parse failed +- [materialize](../basic_json_view/materialize.md) - build the `BasicJsonType` value of a subtree + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/shrink_to_fit.md b/docs/mkdocs/docs/api/basic_json_document/shrink_to_fit.md new file mode 100644 index 000000000..f8f8a4312 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/shrink_to_fit.md @@ -0,0 +1,58 @@ +# nlohmann::basic_json_document::shrink_to_fit + +```cpp +void shrink_to_fit(); +``` + +Releases capacity that is no longer needed, both of the node index and of the buffer for decoded strings (strings that +contained escape sequences), e.g. after [`read()`](read.md) replaced a large document with a much smaller one. Like +`#!cpp std::vector::shrink_to_fit()`, this is a non-binding request: the library may keep more capacity than strictly +necessary. + +## Exception safety + +Strong guarantee: if an exception is thrown, there are no changes to the document. + +## Exceptions + +May throw `#!cpp std::bad_alloc` if the reallocation fails; on exception, the document is unchanged. + +## Complexity + +Linear in [`node_count()`](node_count.md) plus the length of the decoded strings. + +## Notes + +!!! warning "Invalidates views" + + Unlike moving the document, `shrink_to_fit()` **invalidates every view taken from this document before the + call**, including a previously obtained [`root()`](root.md): the node index is moved into a new, smaller + allocation, and the old one is freed. Take a fresh view from [`root()`](root.md) after calling this function. + +This is unlike `#!cpp std::vector::shrink_to_fit()`, which promises nothing about validity but in practice often +leaves iterators alone when it did not need to reallocate; here, an implementation that avoids a reallocation when +possible would be an internal optimization only, not a guarantee to rely on. + +## Examples + +??? example + + ```cpp + --8<-- "examples/basic_json_document__shrink_to_fit.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__shrink_to_fit.output" + ``` + +## See also + +- [node_count](node_count.md) - the number of index entries +- [memory_usage](memory_usage.md) - the number of bytes held by the document +- [root](root.md) - the view of the root value + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/source.md b/docs/mkdocs/docs/api/basic_json_document/source.md new file mode 100644 index 000000000..dcefb88a9 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/source.md @@ -0,0 +1,48 @@ +# nlohmann::basic_json_document::source + +```cpp +view_type::string_view_t source() const noexcept; +``` + +Returns the parsed text, whether it is borrowed from the caller or owned by the document. + +## Return value + +A `#!cpp string_view_t` (`#!cpp std::string_view` on C++17 and newer) over the parsed text, or an empty one if the +document is [discarded](is_discarded.md). + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +For a borrowed document, `source()` points directly into the caller's buffer, so it is only valid while that buffer +is; see [`owns_source`](owns_source.md). + +## Examples + +??? example + + ```cpp + --8<-- "examples/basic_json_document__source.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__source.output" + ``` + +## See also + +- [owns_source](owns_source.md) - return whether the document holds its own copy of the text +- [source_offset](../basic_json_view/source_offset.md) - byte offset of a value in the source text + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md b/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md new file mode 100644 index 000000000..143ea2bea --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/basic_json_view.md @@ -0,0 +1,51 @@ +# nlohmann::basic_json_view::basic_json_view + +```cpp +basic_json_view() noexcept = default; +``` + +Creates an invalid (discarded) view: [`type()`](type.md) is `#!cpp value_t::discarded`, +[`is_discarded()`](is_discarded.md) is `#!cpp true`, and `#!cpp explicit operator bool()` is `#!cpp false`. + +This is the only constructor a caller can use directly. Every other view is obtained from a +[`basic_json_document`](../basic_json_document/index.md), via [`root()`](../basic_json_document/root.md) or (once +element access is added) from navigating into a container. + +## Exception safety + +No-throw guarantee: this constructor never throws exceptions. + +## Complexity + +Constant. + +## Notes + +`basic_json_view` is trivially copyable (it holds two pointers), so a default-constructed view can be used as a +placeholder for "no value yet" and later be assigned a real view. + +## Examples + +??? example + + The example below shows the default constructor and that a `basic_json_view` is a small, copyable handle. + + ```cpp + --8<-- "examples/basic_json_view__basic_json_view.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__basic_json_view.output" + ``` + +## See also + +- [is_discarded](is_discarded.md) - return whether the view is invalid +- [operator bool](operator_bool.md) - return whether the view refers to a value +- [root](../basic_json_document/root.md) - the view of a document's root value + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/empty.md b/docs/mkdocs/docs/api/basic_json_view/empty.md new file mode 100644 index 000000000..0d57fe9e7 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/empty.md @@ -0,0 +1,61 @@ +# nlohmann::basic_json_view::empty + +```cpp +bool empty() const noexcept; +``` + +Checks whether [`size()`](size.md) is `0`, as [`BasicJsonType::empty()`](../basic_json/empty.md) would for the same +value. + +## Return value + +The return value depends on the type and is defined as follows: + +| Value type | return value | +|----------------------|-----------------| +| null | `#!cpp true` | +| discarded | `#!cpp true` | +| boolean | `#!cpp false` | +| string | `#!cpp false` | +| number | `#!cpp false` | +| object | `#!cpp object_t::empty()` | +| array | `#!cpp array_t::empty()` | + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +As for [`BasicJsonType::empty()`](../basic_json/empty.md), this does not return whether a string value is empty -- it +is `#!cpp false` for any string, regardless of its length. + +## Examples + +??? example + + The example below uses [`size()`](size.md) and `empty()` to decide whether a parsed message is worth acting on, + without materializing it into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__size_empty.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__size_empty.output" + ``` + +## See also + +- [size](size.md) - return the number of elements +- [`BasicJsonType::empty`](../basic_json/empty.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/index.md b/docs/mkdocs/docs/api/basic_json_view/index.md new file mode 100644 index 000000000..564406667 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/index.md @@ -0,0 +1,82 @@ +# nlohmann::basic_json_view + +Defined in header `` + +```cpp +template +class basic_json_view; +``` + +A read-only handle to one value of a [`basic_json_document`](../basic_json_document/index.md): two pointers (a pointer +to the document and a pointer into its index), trivially copyable. A view is valid as long as + +- the document is alive, +- the document has not been re-parsed with [`read()`](../basic_json_document/read.md) (or + [`parse()`](../basic_json_document/parse.md) into it) or shrunk with + [`shrink_to_fit()`](../basic_json_document/shrink_to_fit.md) since the view was taken, and +- if the document borrows its source text, that text is still alive. + +Moving the document itself does not invalidate its views: the index is heap-allocated independently of the +`basic_json_document` object. + +`basic_json_view` provides the read-only part of the `BasicJsonType` interface: the type-inspection functions, and +[`materialize()`](materialize.md) to build the `BasicJsonType` value of a subtree on demand. It does not (yet) provide +element access, iteration, `get()`, JSON Pointer support, `dump()`, or comparison. + +## Template parameters + +`BasicJsonType` +: a specialization of [`basic_json`](../basic_json/index.md), matching the + [`basic_json_document`](../basic_json_document/index.md) the view was taken from. + +## Specializations + +- [**json_view**](../json_view.md) - views of a [`json_document`](../json_document.md) +- [**ordered_json_view**](../ordered_json_view.md) - views of an [`ordered_json_document`](../ordered_json_document.md) + +## Member types + +- **value_t** - the JSON type enumeration, see [`basic_json::value_t`](../basic_json/value_t.md) +- **string_t**, **number_integer_t**, **number_unsigned_t**, **number_float_t**, **json_pointer** - the corresponding + member types of `BasicJsonType` +- **size_type** - `#!cpp std::size_t` +- **string_view_t** - `#!cpp std::string_view` on C++17 and newer, a minimal internal substitute otherwise + +## Member functions + +- [(constructor)](basic_json_view.md) + +### Object inspection + +- [**type**](type.md) - return the type of the value +- [**is_null**](is_null.md) - return whether the value is null +- [**is_boolean**](is_boolean.md) - return whether the value is a boolean +- [**is_number**](is_number.md) - return whether the value is a number +- [**is_number_integer**](is_number_integer.md) - return whether the value is an integer number +- [**is_number_unsigned**](is_number_unsigned.md) - return whether the value is an unsigned integer number +- [**is_number_float**](is_number_float.md) - return whether the value is a floating-point number +- [**is_string**](is_string.md) - return whether the value is a string +- [**is_array**](is_array.md) - return whether the value is an array +- [**is_object**](is_object.md) - return whether the value is an object +- [**is_binary**](is_binary.md) - return whether the value is a binary array (always `#!cpp false`) +- [**is_primitive**](is_primitive.md) - return whether the type is primitive +- [**is_structured**](is_structured.md) - return whether the type is structured +- [**is_discarded**](is_discarded.md) - return whether the view is invalid +- [**operator bool**](operator_bool.md) - return whether the view refers to a value + +### Capacity + +- [**size**](size.md) - return the number of elements +- [**empty**](empty.md) - return whether the value has no elements + +### Conversion + +- [**materialize**](materialize.md) - build the `BasicJsonType` value of this subtree + +### Source access + +- [**source_offset**](source_offset.md) - byte offset of this value in the document's source text + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_array.md b/docs/mkdocs/docs/api/basic_json_view/is_array.md new file mode 100644 index 000000000..a33176d2b --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_array.md @@ -0,0 +1,46 @@ +# nlohmann::basic_json_view::is_array + +```cpp +bool is_array() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is an array. + +## Return value + +`#!cpp true` if the type is an array, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_structured](is_structured.md) - return whether the type is structured +- [size](size.md), [empty](empty.md) - the number of elements, and whether there are none +- [`BasicJsonType::is_array`](../basic_json/is_array.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_binary.md b/docs/mkdocs/docs/api/basic_json_view/is_binary.md new file mode 100644 index 000000000..829a1fbb3 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_binary.md @@ -0,0 +1,46 @@ +# nlohmann::basic_json_view::is_binary + +```cpp +bool is_binary() const noexcept; +``` + +This function always returns `#!cpp false`: a JSON text has no binary values, so a view can never refer to one. The +function exists for interface parity with [`BasicJsonType::is_binary`](../basic_json/is_binary.md). + +## Return value + +`#!cpp false`, always. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [type](type.md) - return the type of the value +- [`BasicJsonType::is_binary`](../basic_json/is_binary.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_boolean.md b/docs/mkdocs/docs/api/basic_json_view/is_boolean.md new file mode 100644 index 000000000..148fe2a61 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_boolean.md @@ -0,0 +1,45 @@ +# nlohmann::basic_json_view::is_boolean + +```cpp +bool is_boolean() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is a boolean. + +## Return value + +`#!cpp true` if the type is a boolean, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [type](type.md) - return the type of the value +- [`BasicJsonType::is_boolean`](../basic_json/is_boolean.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_discarded.md b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md new file mode 100644 index 000000000..4cc11077d --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_discarded.md @@ -0,0 +1,55 @@ +# nlohmann::basic_json_view::is_discarded + +```cpp +bool is_discarded() const noexcept; +``` + +Returns whether this view is invalid, i.e. does not refer to a value. This is the case for a default-constructed +view (see [(constructor)](basic_json_view.md)), and for [`root()`](../basic_json_document/root.md) of a document +that is itself [discarded](../basic_json_document/is_discarded.md) -- in particular, the root of a failed +[`parse()`](../basic_json_document/parse.md) with `allow_exceptions` set to `#!cpp false`. + +## Return value + +`#!cpp true` if the view is discarded, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +`#!cpp v.is_discarded()` and `#!cpp !static_cast(v)` are equivalent; use whichever reads better at the call +site. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [operator bool](operator_bool.md) - return whether the view refers to a value +- [(constructor)](basic_json_view.md) - the default constructor creates a discarded view +- [is_discarded (basic_json_document)](../basic_json_document/is_discarded.md) - return whether the last parse failed +- [`BasicJsonType::is_discarded`](../basic_json/is_discarded.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_null.md b/docs/mkdocs/docs/api/basic_json_view/is_null.md new file mode 100644 index 000000000..fffaa836b --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_null.md @@ -0,0 +1,45 @@ +# nlohmann::basic_json_view::is_null + +```cpp +bool is_null() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is `#!json null`. + +## Return value + +`#!cpp true` if the type is `#!json null`, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [type](type.md) - return the type of the value +- [`BasicJsonType::is_null`](../basic_json/is_null.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_number.md b/docs/mkdocs/docs/api/basic_json_view/is_number.md new file mode 100644 index 000000000..158734364 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_number.md @@ -0,0 +1,47 @@ +# nlohmann::basic_json_view::is_number + +```cpp +bool is_number() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is a number, i.e. an integer, unsigned integer, or floating-point value. It is defined as `#!cpp is_number_integer() || is_number_float()`. + +## Return value + +`#!cpp true` if the type is a number, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_number_integer](is_number_integer.md) - return whether the value is an integer or unsigned integer number +- [is_number_unsigned](is_number_unsigned.md) - return whether the value is an unsigned integer number +- [is_number_float](is_number_float.md) - return whether the value is a floating-point number +- [`BasicJsonType::is_number`](../basic_json/is_number.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_number_float.md b/docs/mkdocs/docs/api/basic_json_view/is_number_float.md new file mode 100644 index 000000000..52bb43f20 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_number_float.md @@ -0,0 +1,51 @@ +# nlohmann::basic_json_view::is_number_float + +```cpp +bool is_number_float() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is a floating-point number. + +## Return value + +`#!cpp true` if the type is `#!cpp value_t::number_float`, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +As for [`BasicJsonType::parse`](../basic_json/parse.md), an integer literal that does not fit into the 64-bit +integer type is classified as a floating-point number, so `is_number_float()` can be `#!cpp true` even for an +integer-looking token in the source text. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_number](is_number.md) - return whether the value is a number +- [`BasicJsonType::is_number_float`](../basic_json/is_number_float.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_number_integer.md b/docs/mkdocs/docs/api/basic_json_view/is_number_integer.md new file mode 100644 index 000000000..fda0c29c9 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_number_integer.md @@ -0,0 +1,51 @@ +# nlohmann::basic_json_view::is_number_integer + +```cpp +bool is_number_integer() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is an integer or unsigned integer number. + +## Return value + +`#!cpp true` if the type is `#!cpp value_t::number_integer` or `#!cpp value_t::number_unsigned`, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +As for [`BasicJsonType::is_number_integer`](../basic_json/is_number_integer.md), this includes unsigned integer +values; use [`is_number_unsigned`](is_number_unsigned.md) to test for those specifically. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_number](is_number.md) - return whether the value is a number +- [is_number_unsigned](is_number_unsigned.md) - return whether the value is an unsigned integer number +- [`BasicJsonType::is_number_integer`](../basic_json/is_number_integer.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_number_unsigned.md b/docs/mkdocs/docs/api/basic_json_view/is_number_unsigned.md new file mode 100644 index 000000000..d2cf10ebe --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_number_unsigned.md @@ -0,0 +1,45 @@ +# nlohmann::basic_json_view::is_number_unsigned + +```cpp +bool is_number_unsigned() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is an unsigned integer number. + +## Return value + +`#!cpp true` if the type is `#!cpp value_t::number_unsigned`, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_number_integer](is_number_integer.md) - return whether the value is an integer or unsigned integer number +- [`BasicJsonType::is_number_unsigned`](../basic_json/is_number_unsigned.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_object.md b/docs/mkdocs/docs/api/basic_json_view/is_object.md new file mode 100644 index 000000000..5421aeeb1 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_object.md @@ -0,0 +1,46 @@ +# nlohmann::basic_json_view::is_object + +```cpp +bool is_object() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is an object. + +## Return value + +`#!cpp true` if the type is an object, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_structured](is_structured.md) - return whether the type is structured +- [size](size.md), [empty](empty.md) - the number of elements, and whether there are none +- [`BasicJsonType::is_object`](../basic_json/is_object.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_primitive.md b/docs/mkdocs/docs/api/basic_json_view/is_primitive.md new file mode 100644 index 000000000..a28b478f0 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_primitive.md @@ -0,0 +1,46 @@ +# nlohmann::basic_json_view::is_primitive + +```cpp +bool is_primitive() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is primitive, i.e. `#!json null`, a boolean, a number, or a string. It is defined as `#!cpp is_null() || is_string() || is_boolean() || is_number()`. + +## Return value + +`#!cpp true` if the type is primitive, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_structured](is_structured.md) - return whether the type is structured (the complement of this function, for a + non-discarded view) +- [`BasicJsonType::is_primitive`](../basic_json/is_primitive.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_string.md b/docs/mkdocs/docs/api/basic_json_view/is_string.md new file mode 100644 index 000000000..c40c8ab4b --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_string.md @@ -0,0 +1,45 @@ +# nlohmann::basic_json_view::is_string + +```cpp +bool is_string() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is a string. + +## Return value + +`#!cpp true` if the type is a string, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [source_offset](source_offset.md) - byte offset of this value in the document's source text +- [`BasicJsonType::is_string`](../basic_json/is_string.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/is_structured.md b/docs/mkdocs/docs/api/basic_json_view/is_structured.md new file mode 100644 index 000000000..de5dd5f7e --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/is_structured.md @@ -0,0 +1,46 @@ +# nlohmann::basic_json_view::is_structured + +```cpp +bool is_structured() const noexcept; +``` + +This function returns `#!cpp true` if and only if the value is structured, i.e. an array or an object. It is defined as `#!cpp is_array() || is_object()`. + +## Return value + +`#!cpp true` if the type is structured, `#!cpp false` otherwise. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_primitive](is_primitive.md) - return whether the type is primitive +- [size](size.md), [empty](empty.md) - the number of elements, and whether there are none +- [`BasicJsonType::is_structured`](../basic_json/is_structured.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/materialize.md b/docs/mkdocs/docs/api/basic_json_view/materialize.md new file mode 100644 index 000000000..225af3c9a --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/materialize.md @@ -0,0 +1,69 @@ +# nlohmann::basic_json_view::materialize + +```cpp +BasicJsonType materialize() const; +``` + +Builds the `BasicJsonType` value of this subtree: the value [`BasicJsonType::parse()`](../basic_json/parse.md) would +have produced for the same source text, allocated for the first time by this call. + +## Return value + +The `BasicJsonType` value of this subtree, or a discarded `BasicJsonType` value (`#!cpp BasicJsonType(value_t::discarded)`) +if the view is [discarded](is_discarded.md). + +## Exception safety + +Strong guarantee: if an exception is thrown, there are no changes to the view or the document it refers to (nothing +about either is mutated by this function). + +## Exceptions + +May throw `#!cpp std::bad_alloc` (via `BasicJsonType`'s allocator) if constructing the result fails. + +## Complexity + +Linear in the size of the subtree. + +## Notes + +`materialize()` replays the subtree through the same SAX builder [`BasicJsonType::parse()`](../basic_json/parse.md) +uses internally, so the result matches it exactly -- including, for an object, that a repeated key keeps only its +last value. The replay is iterative, so it is not limited by the call stack the way a naive recursive conversion +would be; the JSON nesting depth is limited only by available memory, as for `BasicJsonType::parse()` itself. + +Unlike parsing with [`JSON_DIAGNOSTIC_POSITIONS`](../macros/json_diagnostic_positions.md) enabled, the values +produced by `materialize()` do not carry source positions: there is no lexer run during the replay to record them. + +Calling `materialize()` on the same view repeatedly builds a new, independent `BasicJsonType` value each time; it +never caches the result. + +## Examples + +??? example + + The example below skips messages that are not useful -- a discarded value, or an empty array -- using only + [`is_array()`](is_array.md) and [`empty()`](empty.md), and calls `materialize()` only for the messages that are + actually used, so no `BasicJsonType` value (and none of its per-element allocations) is ever built for the + skipped ones. + + ```cpp + --8<-- "examples/basic_json_view__materialize.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__materialize.output" + ``` + +## See also + +- [root](../basic_json_document/root.md) - the view of a document's root value +- [`BasicJsonType::parse`](../basic_json/parse.md) - build a `BasicJsonType` value directly from a JSON text +- [`JSON_DIAGNOSTIC_POSITIONS`](../macros/json_diagnostic_positions.md) - source positions on parsed values (not + produced by `materialize()`) + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/operator_bool.md b/docs/mkdocs/docs/api/basic_json_view/operator_bool.md new file mode 100644 index 000000000..bc13c100c --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/operator_bool.md @@ -0,0 +1,46 @@ +# nlohmann::basic_json_view::operator bool + +```cpp +explicit operator bool() const noexcept; +``` + +Returns whether this view refers to a value, i.e. the negation of [`is_discarded()`](is_discarded.md). Being +`#!cpp explicit`, this conversion is only considered in a boolean context (`#!cpp if (v)`, `#!cpp !v`, `#!cpp v && +...`), not for implicit conversions to other types. + +## Return value + +`#!cpp true` if the view refers to a value, `#!cpp false` if it is [discarded](is_discarded.md). + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_discarded](is_discarded.md) - return whether the view is invalid + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/size.md b/docs/mkdocs/docs/api/basic_json_view/size.md new file mode 100644 index 000000000..cd354d2f3 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/size.md @@ -0,0 +1,60 @@ +# nlohmann::basic_json_view::size + +```cpp +size_type size() const noexcept; +``` + +Returns the number of elements, as [`BasicJsonType::size()`](../basic_json/size.md) would for the same value. + +## Return value + +The return value depends on the type and is defined as follows: + +| Value type | return value | +|----------------------|-----------------------------| +| null | `0` | +| discarded | `0` | +| boolean | `1` | +| string | `1` | +| number | `1` | +| object | number of key/value pairs | +| array | number of elements | + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant: for an object or array, the element count is stored in the index, not counted on demand. + +## Notes + +As for [`BasicJsonType::size()`](../basic_json/size.md), this does not return the length of a string value -- it is +`1` for a string, regardless of its length. + +## Examples + +??? example + + The example below uses `size()` and [`empty()`](empty.md) to decide whether a parsed message is worth acting on, + without materializing it into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__size_empty.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__size_empty.output" + ``` + +## See also + +- [empty](empty.md) - return whether the value has no elements +- [`BasicJsonType::size`](../basic_json/size.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/source_offset.md b/docs/mkdocs/docs/api/basic_json_view/source_offset.md new file mode 100644 index 000000000..5f7e06221 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/source_offset.md @@ -0,0 +1,55 @@ +# nlohmann::basic_json_view::source_offset + +```cpp +std::size_t source_offset() const noexcept; +``` + +Returns the byte offset of this value in the document's [`source()`](../basic_json_document/source.md) text, without +materializing anything. + +## Return value + +- For a string with no escapes, a number, `#!json true`/`#!json false`/`#!json null`, an array, or an object: the + byte offset of the first byte of the value's token (for a string: the first byte after the opening quote) in + [`source()`](../basic_json_document/source.md). +- `#!cpp static_cast(-1)` for a [discarded](is_discarded.md) view, and for a string that contains + escapes -- such a string was decoded once into the document's own buffer, so there is no single byte range in + `source()` left to point at. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +This is a raw offset, not a length: the API does not (yet) expose how many bytes the token occupies in the source +text, so `source_offset()` alone is enough to report *where* a value came from (for an error message, for syntax +highlighting, ...) but not to slice its exact text back out of [`source()`](../basic_json_document/source.md) for a +string, whose token length is not the same as its decoded [`size()`](size.md). + +## Examples + +??? example + + ```cpp + --8<-- "examples/basic_json_view__source_offset.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__source_offset.output" + ``` + +## See also + +- [source](../basic_json_document/source.md) - the parsed text +- [is_string](is_string.md) - return whether the value is a string + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_view/type.md b/docs/mkdocs/docs/api/basic_json_view/type.md new file mode 100644 index 000000000..17dbc3c34 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_view/type.md @@ -0,0 +1,53 @@ +# nlohmann::basic_json_view::type + +```cpp +value_t type() const noexcept; +``` + +Returns the type of the value this view refers to, as a value from the [`value_t`](../basic_json/value_t.md) +enumeration -- the same enumeration [`BasicJsonType::type()`](../basic_json/type.md) uses. + +## Return value + +The type of the value; `#!cpp value_t::discarded` for a [discarded](is_discarded.md) view. + +## Exception safety + +No-throw guarantee: this function never throws exceptions. + +## Complexity + +Constant. + +## Notes + +Unlike [`BasicJsonType::type()`](../basic_json/type.md), this function can never return `#!cpp value_t::binary`: a +JSON text has no binary values, so `type()` only distinguishes the eight ordinary JSON value types (plus +`#!cpp discarded`). + +## Examples + +??? example + + The example below classifies several parsed documents by the type of their root value, without materializing any + of them into a `BasicJsonType` value. + + ```cpp + --8<-- "examples/basic_json_view__type_predicates.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_view__type_predicates.output" + ``` + +## See also + +- [is_null](is_null.md), [is_boolean](is_boolean.md), [is_number](is_number.md), [is_string](is_string.md), + [is_array](is_array.md), [is_object](is_object.md) - type-specific predicates built on `type()` +- [`BasicJsonType::type`](../basic_json/type.md) - the corresponding function of `basic_json` + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/json_document.md b/docs/mkdocs/docs/api/json_document.md new file mode 100644 index 000000000..ea365badc --- /dev/null +++ b/docs/mkdocs/docs/api/json_document.md @@ -0,0 +1,35 @@ +# nlohmann::json_document + +Defined in header `` + +```cpp +using json_document = basic_json_document; +``` + +This type is a [`basic_json_document`](basic_json_document/index.md) of the default [`json`](json.md) +specialization. + +## Examples + +??? example + + The example below demonstrates how to use the type `nlohmann::json_document`. + + ```cpp + --8<-- "examples/json_document.cpp" + ``` + + Output: + + ```json + --8<-- "examples/json_document.output" + ``` + +## See also + +- [json_view](json_view.md) - a view of a value of a `json_document` +- [ordered_json_document](ordered_json_document.md) - the corresponding document for `ordered_json` + +## Version history + +Since version 3.13.0. diff --git a/docs/mkdocs/docs/api/json_view.md b/docs/mkdocs/docs/api/json_view.md new file mode 100644 index 000000000..819dfe501 --- /dev/null +++ b/docs/mkdocs/docs/api/json_view.md @@ -0,0 +1,34 @@ +# nlohmann::json_view + +Defined in header `` + +```cpp +using json_view = basic_json_view; +``` + +This type is a [`basic_json_view`](basic_json_view/index.md) of a value of a [`json_document`](json_document.md). + +## Examples + +??? example + + The example below demonstrates how to use the type `nlohmann::json_view`. + + ```cpp + --8<-- "examples/json_view.cpp" + ``` + + Output: + + ```json + --8<-- "examples/json_view.output" + ``` + +## See also + +- [json_document](json_document.md) - the document type this view refers into +- [ordered_json_view](ordered_json_view.md) - the corresponding view for `ordered_json_document` + +## Version history + +Since version 3.13.0. diff --git a/docs/mkdocs/docs/api/ordered_json_document.md b/docs/mkdocs/docs/api/ordered_json_document.md new file mode 100644 index 000000000..dd6760ead --- /dev/null +++ b/docs/mkdocs/docs/api/ordered_json_document.md @@ -0,0 +1,38 @@ +# nlohmann::ordered_json_document + +Defined in header `` + +```cpp +using ordered_json_document = basic_json_document; +``` + +This type is a [`basic_json_document`](basic_json_document/index.md) of the [`ordered_json`](ordered_json.md) +specialization: [`materialize()`](basic_json_view/materialize.md) on one of its views preserves the insertion order +of object keys, instead of sorting them like [`json_document`](json_document.md) does. + +## Examples + +??? example + + The example below demonstrates how `ordered_json_document` preserves the insertion order of object keys when + materializing. + + ```cpp + --8<-- "examples/ordered_json_document.cpp" + ``` + + Output: + + ```json + --8<-- "examples/ordered_json_document.output" + ``` + +## See also + +- [ordered_json_view](ordered_json_view.md) - a view of a value of an `ordered_json_document` +- [json_document](json_document.md) - the corresponding document for the default `json` specialization +- [Object Order](../features/object_order.md) + +## Version history + +Since version 3.13.0. diff --git a/docs/mkdocs/docs/api/ordered_json_view.md b/docs/mkdocs/docs/api/ordered_json_view.md new file mode 100644 index 000000000..9800b0752 --- /dev/null +++ b/docs/mkdocs/docs/api/ordered_json_view.md @@ -0,0 +1,35 @@ +# nlohmann::ordered_json_view + +Defined in header `` + +```cpp +using ordered_json_view = basic_json_view; +``` + +This type is a [`basic_json_view`](basic_json_view/index.md) of a value of an +[`ordered_json_document`](ordered_json_document.md). + +## Examples + +??? example + + The example below demonstrates how to use the type `nlohmann::ordered_json_view`. + + ```cpp + --8<-- "examples/ordered_json_view.cpp" + ``` + + Output: + + ```json + --8<-- "examples/ordered_json_view.output" + ``` + +## See also + +- [ordered_json_document](ordered_json_document.md) - the document type this view refers into +- [json_view](json_view.md) - the corresponding view for `json_document` + +## Version history + +Since version 3.13.0. diff --git a/docs/mkdocs/docs/examples/basic_json_document__accept.cpp b/docs/mkdocs/docs/examples/basic_json_document__accept.cpp new file mode 100644 index 000000000..675fa823f --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__accept.cpp @@ -0,0 +1,19 @@ +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // accept() behaves exactly like basic_json::accept(): the same inputs are + // accepted or rejected, with the same ignore_comments/ignore_trailing_commas + // options + std::cout << json_document::accept(R"({"a": 1})") << '\n'; + std::cout << json_document::accept(R"({"a": 1,})") << '\n'; // trailing comma: rejected by default + std::cout << json_document::accept(R"({"a": 1,})", false, true) << '\n'; // ignore_trailing_commas + + std::cout << (json_document::accept(R"({"a": 1})") == nlohmann::json::accept(R"({"a": 1})")) << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__accept.output b/docs/mkdocs/docs/examples/basic_json_document__accept.output new file mode 100644 index 000000000..271474188 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__accept.output @@ -0,0 +1,4 @@ +true +false +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_document__basic_json_document.cpp b/docs/mkdocs/docs/examples/basic_json_document__basic_json_document.cpp new file mode 100644 index 000000000..0c185f7b8 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__basic_json_document.cpp @@ -0,0 +1,23 @@ +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // the default constructor creates an empty (discarded) document + json_document empty; + std::cout << empty.is_discarded() << '\n'; + + // json_document is move-only: parse() itself returns by value (moved out), + // and a document can be moved again, e.g. into a container + json_document doc = json_document::parse(R"({"a": 1})"); + json_document moved = std::move(doc); + std::cout << moved.root().is_object() << '\n'; + + // copying is disabled at compile time: + // json_document another = moved; // does not compile +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__basic_json_document.output b/docs/mkdocs/docs/examples/basic_json_document__basic_json_document.output new file mode 100644 index 000000000..bb101b641 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__basic_json_document.output @@ -0,0 +1,2 @@ +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_document__is_discarded.cpp b/docs/mkdocs/docs/examples/basic_json_document__is_discarded.cpp new file mode 100644 index 000000000..d64d37d23 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__is_discarded.cpp @@ -0,0 +1,19 @@ +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // with allow_exceptions == false, a parse error produces a discarded + // document instead of throwing -- exactly like basic_json::parse() + json_document doc = json_document::parse(R"({"a": )", /* allow_exceptions */ false); + std::cout << doc.is_discarded() << '\n'; + std::cout << doc.root().is_discarded() << '\n'; + + // a successful parse is never discarded + json_document ok = json_document::parse(R"({"a": 1})", false); + std::cout << ok.is_discarded() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__is_discarded.output b/docs/mkdocs/docs/examples/basic_json_document__is_discarded.output new file mode 100644 index 000000000..9e8a46acf --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__is_discarded.output @@ -0,0 +1,3 @@ +true +true +false diff --git a/docs/mkdocs/docs/examples/basic_json_document__memory_usage.cpp b/docs/mkdocs/docs/examples/basic_json_document__memory_usage.cpp new file mode 100644 index 000000000..efa235310 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__memory_usage.cpp @@ -0,0 +1,28 @@ +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // long enough that copying it needs a real (heap) allocation, so the + // comparison below does not depend on the standard library's small + // string optimization threshold + const std::string text = std::string(200, ' ') + "[1, 2, 3, 4, 5]"; + + // memory_usage() is not portable across platforms/allocators/compilers, so + // compare it relatively instead of printing the raw byte count + json_document borrowed = json_document::parse(text); + json_document owned = json_document::parse_copy(text); + + // the owned document additionally stores its own copy of the source text + std::cout << (owned.memory_usage() > borrowed.memory_usage()) << '\n'; + + // a document with more values needs a larger index + json_document small = json_document::parse(std::string("[1]")); + json_document large = json_document::parse(std::string("[1, 2, 3, 4, 5, 6, 7, 8, 9, 10]")); + std::cout << (large.memory_usage() > small.memory_usage()) << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__memory_usage.output b/docs/mkdocs/docs/examples/basic_json_document__memory_usage.output new file mode 100644 index 000000000..bb101b641 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__memory_usage.output @@ -0,0 +1,2 @@ +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_document__node_count.cpp b/docs/mkdocs/docs/examples/basic_json_document__node_count.cpp new file mode 100644 index 000000000..d6ae5f89b --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__node_count.cpp @@ -0,0 +1,15 @@ +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + // node_count() is the size of the flat index: one 16-byte node per value, + // plus one per object key (values and keys are all the index stores) + json_document scalar = json_document::parse("42"); + std::cout << scalar.node_count() << '\n'; + + json_document doc = json_document::parse(R"({"a": 1, "b": [1, 2]})"); + std::cout << doc.node_count() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__node_count.output b/docs/mkdocs/docs/examples/basic_json_document__node_count.output new file mode 100644 index 000000000..fea32e7d8 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__node_count.output @@ -0,0 +1,2 @@ +1 +7 diff --git a/docs/mkdocs/docs/examples/basic_json_document__owns_source.cpp b/docs/mkdocs/docs/examples/basic_json_document__owns_source.cpp new file mode 100644 index 000000000..8f442ea13 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__owns_source.cpp @@ -0,0 +1,24 @@ +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + std::string text = R"({"a": 1})"; + + // borrowed: the document only points into `text`; `text` must outlive it + json_document borrowed = json_document::parse(text); + std::cout << borrowed.owns_source() << '\n'; + + // owned: parse_copy() always takes its own copy + json_document copied = json_document::parse_copy(text); + std::cout << copied.owns_source() << '\n'; + + // owned: an rvalue std::string is moved in, not copied, but still owned + json_document moved_in = json_document::parse(std::string(text)); + std::cout << moved_in.owns_source() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__owns_source.output b/docs/mkdocs/docs/examples/basic_json_document__owns_source.output new file mode 100644 index 000000000..dac32ecf1 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__owns_source.output @@ -0,0 +1,3 @@ +false +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_document__parse.cpp b/docs/mkdocs/docs/examples/basic_json_document__parse.cpp new file mode 100644 index 000000000..4eba7350d --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__parse.cpp @@ -0,0 +1,33 @@ +#include +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // an lvalue std::string is BORROWED: the document only stores a pointer + // into `text`, so `text` must outlive `borrowed` + std::string text = R"({"count": 3})"; + json_document borrowed = json_document::parse(text); + std::cout << borrowed.owns_source() << '\n'; // false + + // an rvalue std::string is MOVED into the document -- no copy of the text + json_document owned = json_document::parse(std::string(R"({"count": 3})")); + std::cout << owned.owns_source() << '\n'; // true + + // errors are identical to basic_json::parse: same exception id, message, + // and position, because the library parser runs on the same bytes on a + // failing input + try + { + static_cast(json_document::parse(R"({"count": )")); + } + catch (const nlohmann::json::parse_error& e) + { + std::cout << e.id << '\n'; + } +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__parse.output b/docs/mkdocs/docs/examples/basic_json_document__parse.output new file mode 100644 index 000000000..99fd51a73 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__parse.output @@ -0,0 +1,3 @@ +false +true +101 diff --git a/docs/mkdocs/docs/examples/basic_json_document__parse_copy.cpp b/docs/mkdocs/docs/examples/basic_json_document__parse_copy.cpp new file mode 100644 index 000000000..16d3ce939 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__parse_copy.cpp @@ -0,0 +1,22 @@ +#include +#include + +using json_document = nlohmann::json_document; + +json_document parse_from_temporary_buffer() +{ + char buffer[] = R"({"a": 1})"; + // parse() would borrow `buffer`, which is about to go out of scope; + // parse_copy() takes its own copy instead, so the returned document does + // not depend on `buffer` afterward + return json_document::parse_copy(buffer); +} + +int main() +{ + std::cout << std::boolalpha; + + json_document doc = parse_from_temporary_buffer(); + std::cout << doc.owns_source() << '\n'; + std::cout << doc.root().is_object() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__parse_copy.output b/docs/mkdocs/docs/examples/basic_json_document__parse_copy.output new file mode 100644 index 000000000..bb101b641 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__parse_copy.output @@ -0,0 +1,2 @@ +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_document__parse_iterator_pair.cpp b/docs/mkdocs/docs/examples/basic_json_document__parse_iterator_pair.cpp new file mode 100644 index 000000000..6d446bd67 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__parse_iterator_pair.cpp @@ -0,0 +1,22 @@ +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // a JSON value embedded in a larger, non-null-terminated buffer (e.g. a + // slice received over the network) + std::vector buffer = {'[', '1', ',', '2', ']', 'j', 'u', 'n', 'k'}; + + // a pointer pair is BORROWED, exactly like a byte container lvalue: the + // document points into the buffer. (From C++20 on, std::vector + // iterators are borrowed as well; before, they are copied.) + const char* first = buffer.data(); + json_document doc = json_document::parse(first, first + 5); + std::cout << doc.root().is_array() << ' ' << doc.root().size() << '\n'; + std::cout << doc.owns_source() << '\n'; // false: borrowed +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__parse_iterator_pair.output b/docs/mkdocs/docs/examples/basic_json_document__parse_iterator_pair.output new file mode 100644 index 000000000..1ea7a5a0e --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__parse_iterator_pair.output @@ -0,0 +1,2 @@ +true 2 +false diff --git a/docs/mkdocs/docs/examples/basic_json_document__read.cpp b/docs/mkdocs/docs/examples/basic_json_document__read.cpp new file mode 100644 index 000000000..4858a7db7 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__read.cpp @@ -0,0 +1,31 @@ +#include +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + std::vector messages = + { + R"({"id": 1})", R"({"id": 2, "tag": "x"})", R"({"id": 3})" + }; + + // parse into the same document over and over: its node index and decode + // buffer are reused instead of being freed and reallocated for each message + json_document doc; + std::size_t total = 0; + for (const auto& msg : messages) + { + doc.read(msg); + total += doc.root().size(); + } + std::cout << total << '\n'; + + // read() can also change what kind of input is owned/borrowed between calls + doc.read(std::string(R"({"owned": true})")); + std::cout << doc.owns_source() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__read.output b/docs/mkdocs/docs/examples/basic_json_document__read.output new file mode 100644 index 000000000..fea15e47f --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__read.output @@ -0,0 +1,2 @@ +4 +true diff --git a/docs/mkdocs/docs/examples/basic_json_document__root.cpp b/docs/mkdocs/docs/examples/basic_json_document__root.cpp new file mode 100644 index 000000000..9789f2566 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__root.cpp @@ -0,0 +1,19 @@ +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + json_document doc = json_document::parse(R"({"greeting": "hi"})"); + std::cout << doc.root().is_object() << '\n'; + + // root() is a cheap handle, not a copy: repeated calls observe the same value + std::cout << (doc.root().type() == doc.root().type()) << '\n'; + + // the root of a failed parse (allow_exceptions == false) is discarded + json_document failed = json_document::parse("{", false); + std::cout << failed.root().is_discarded() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__root.output b/docs/mkdocs/docs/examples/basic_json_document__root.output new file mode 100644 index 000000000..b979d62f4 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__root.output @@ -0,0 +1,3 @@ +true +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_document__shrink_to_fit.cpp b/docs/mkdocs/docs/examples/basic_json_document__shrink_to_fit.cpp new file mode 100644 index 000000000..f2560078a --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__shrink_to_fit.cpp @@ -0,0 +1,44 @@ +#include +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // build a large array (many nodes), then read a small one into the same + // document: the index grown for the large input is still allocated + std::ostringstream big; + big << '['; + for (int i = 0; i < 500; ++i) + { + if (i != 0) + { + big << ','; + } + big << i; + } + big << ']'; + + json_document doc; + doc.read(big.str()); + const std::size_t big_nodes = doc.node_count(); + + doc.read(std::string("[1]")); + std::cout << (doc.node_count() < big_nodes) << '\n'; // far fewer live nodes now + const std::size_t before = doc.memory_usage(); + + // shrink_to_fit() moves the index into a block sized for what is actually + // used. This INVALIDATES every view taken from this document before the + // call (they point into the old, now-freed block) -- take fresh ones from + // root() afterward. + doc.shrink_to_fit(); + const std::size_t after = doc.memory_usage(); + std::cout << (after <= before) << '\n'; + + // a freshly taken view is valid and correct + std::cout << doc.root().is_array() << ' ' << doc.root().size() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__shrink_to_fit.output b/docs/mkdocs/docs/examples/basic_json_document__shrink_to_fit.output new file mode 100644 index 000000000..9e7440e42 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__shrink_to_fit.output @@ -0,0 +1,3 @@ +true +true +true 1 diff --git a/docs/mkdocs/docs/examples/basic_json_document__source.cpp b/docs/mkdocs/docs/examples/basic_json_document__source.cpp new file mode 100644 index 000000000..722cdffaf --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__source.cpp @@ -0,0 +1,20 @@ +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + std::string text = R"({"a": 1})"; + json_document doc = json_document::parse(text); + + // source() is the parsed text, whether borrowed or owned + std::cout << (doc.source().size() == text.size()) << '\n'; + std::cout << std::string(doc.source().data(), doc.source().size()) << '\n'; + + // for a borrowed document, source() points right into the caller's buffer + std::cout << (doc.source().data() == text.data()) << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__source.output b/docs/mkdocs/docs/examples/basic_json_document__source.output new file mode 100644 index 000000000..1570aec3a --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__source.output @@ -0,0 +1,3 @@ +true +{"a": 1} +true diff --git a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp new file mode 100644 index 000000000..af08f22e0 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.cpp @@ -0,0 +1,17 @@ +#include +#include + +int main() +{ + std::cout << std::boolalpha; + + // the default constructor is the only public one: it creates an invalid + // (discarded) view, useful as a "no value yet" placeholder + nlohmann::json_view v; + std::cout << static_cast(v) << ' ' << v.is_discarded() << '\n'; + + // views are trivially copyable handles (two pointers); the document owns + // the actual data + nlohmann::json_view copy = v; + std::cout << static_cast(copy) << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output new file mode 100644 index 000000000..8192d6145 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__basic_json_view.output @@ -0,0 +1,2 @@ +false true +false diff --git a/docs/mkdocs/docs/examples/basic_json_view__materialize.cpp b/docs/mkdocs/docs/examples/basic_json_view__materialize.cpp new file mode 100644 index 000000000..38160b461 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__materialize.cpp @@ -0,0 +1,33 @@ +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + // three incoming messages; skip the ones that are not useful without ever + // building a nlohmann::json value for them + json_document heartbeat = json_document::parse("null"); + json_document empty_batch = json_document::parse("[]"); + json_document batch = json_document::parse(R"([{"id": 1}, {"id": 2}, {"id": 3}])"); + std::array messages = {{&heartbeat, &empty_batch, &batch}}; + + for (const json_document* d : messages) + { + // is_array()/empty() only look at the flat index: a discarded + // heartbeat or an empty batch is never turned into a nlohmann::json + // value, so no per-element allocation happens for them + if (!d->root().is_array() || d->root().empty()) + { + std::cout << "skipped\n"; + continue; + } + + // materialize() replays the subtree through the same SAX builder + // basic_json::parse() uses, so the result is exactly what + // basic_json::parse() would have produced for the same text + nlohmann::json value = d->root().materialize(); + std::cout << value.dump() << '\n'; + } +} diff --git a/docs/mkdocs/docs/examples/basic_json_view__materialize.output b/docs/mkdocs/docs/examples/basic_json_view__materialize.output new file mode 100644 index 000000000..207c875ff --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__materialize.output @@ -0,0 +1,3 @@ +skipped +skipped +[{"id":1},{"id":2},{"id":3}] diff --git a/docs/mkdocs/docs/examples/basic_json_view__size_empty.cpp b/docs/mkdocs/docs/examples/basic_json_view__size_empty.cpp new file mode 100644 index 000000000..36700ef9d --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__size_empty.cpp @@ -0,0 +1,22 @@ +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // decide whether a batch is worth processing before building any + // nlohmann::json value for it + json_document batch = json_document::parse(R"([1, 2, 3, 4, 5])"); + json_document empty_batch = json_document::parse("[]"); + + std::cout << batch.root().empty() << ' ' << batch.root().size() << '\n'; + std::cout << empty_batch.root().empty() << ' ' << empty_batch.root().size() << '\n'; + + // as for basic_json: null has size 0, every other scalar has size 1 + json_document n = json_document::parse("null"); + json_document s = json_document::parse(R"("hi")"); + std::cout << n.root().size() << ' ' << s.root().size() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_view__size_empty.output b/docs/mkdocs/docs/examples/basic_json_view__size_empty.output new file mode 100644 index 000000000..8e7ccf182 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__size_empty.output @@ -0,0 +1,3 @@ +false 5 +true 0 +0 1 diff --git a/docs/mkdocs/docs/examples/basic_json_view__source_offset.cpp b/docs/mkdocs/docs/examples/basic_json_view__source_offset.cpp new file mode 100644 index 000000000..651964f0f --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__source_offset.cpp @@ -0,0 +1,33 @@ +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // source_offset() points into source(): useful to report *where* in the + // original text a value came from (error messages, syntax highlighting, + // forwarding a sub-range verbatim, ...) without materializing it + json_document doc = json_document::parse(R"( 123)"); + auto v = doc.root(); + std::cout << v.source_offset() << ' ' + << std::string(doc.source().data() + v.source_offset(), 3) << '\n'; + + // a string without escapes also stays in the source text + json_document plain = json_document::parse(R"("ab")"); + std::cout << plain.source()[plain.root().source_offset()] << '\n'; + + // a string with escapes is decoded once into the document's own buffer, so + // there is no single byte range in source() to point at: source_offset() + // returns the "not applicable" sentinel + json_document escaped = json_document::parse(R"("a\nb")"); + std::cout << (escaped.root().source_offset() == static_cast(-1)) << '\n'; + + // a discarded view -- default-constructed, or the root of a failed parse + // with allow_exceptions == false -- has no offset either + json_document failed = json_document::parse("not json", false); + std::cout << (failed.root().source_offset() == static_cast(-1)) << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_view__source_offset.output b/docs/mkdocs/docs/examples/basic_json_view__source_offset.output new file mode 100644 index 000000000..eae7db157 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__source_offset.output @@ -0,0 +1,4 @@ +2 123 +a +true +true diff --git a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp new file mode 100644 index 000000000..1e3ed62e2 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.cpp @@ -0,0 +1,43 @@ +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // several incoming messages, one json_document per message. type() and + // is_*() only look at the flat index built by parse(); no nlohmann::json + // tree exists yet, and none is built unless materialize() is called + json_document d_null = json_document::parse("null"); + json_document d_bool = json_document::parse("true"); + json_document d_int = json_document::parse("-42"); + json_document d_unsigned = json_document::parse("42"); + json_document d_float = json_document::parse("4.2"); + json_document d_string = json_document::parse(R"("hi")"); + json_document d_array = json_document::parse("[1, 2, 3]"); + json_document d_object = json_document::parse(R"({"a": 1})"); + + std::cout << d_null.root().is_null() << '\n'; + std::cout << d_bool.root().is_boolean() << '\n'; + std::cout << d_int.root().is_number() << ' ' << d_int.root().is_number_integer() << '\n'; + std::cout << d_unsigned.root().is_number_unsigned() << '\n'; + std::cout << d_float.root().is_number_float() << '\n'; + std::cout << d_string.root().is_string() << '\n'; + std::cout << d_array.root().is_array() << ' ' << d_array.root().is_structured() << '\n'; + std::cout << d_object.root().is_object() << ' ' << d_object.root().is_primitive() << '\n'; + + // JSON text can never produce a binary value: is_binary() is always false + std::cout << d_array.root().is_binary() << '\n'; + + // a default-constructed view, and the root of a document that failed to + // parse without exceptions, are both discarded + nlohmann::json_view invalid; + json_document failed = json_document::parse("not json", /* allow_exceptions */ false); + std::cout << static_cast(invalid) << ' ' << invalid.is_discarded() << '\n'; + std::cout << static_cast(failed.root()) << ' ' << failed.root().is_discarded() << '\n'; + + // type() returns the same value_t enumeration as basic_json::type() + std::cout << (d_object.root().type() == nlohmann::json::value_t::object) << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output new file mode 100644 index 000000000..285d31ca0 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_view__type_predicates.output @@ -0,0 +1,12 @@ +true +true +true true +true +true +true +true true +true false +false +false true +false true +true diff --git a/docs/mkdocs/docs/examples/json_document.cpp b/docs/mkdocs/docs/examples/json_document.cpp new file mode 100644 index 000000000..01f2847e2 --- /dev/null +++ b/docs/mkdocs/docs/examples/json_document.cpp @@ -0,0 +1,16 @@ +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // json_document is basic_json_document: same value types + // and containers as the ordinary json specialization + json_document doc = json_document::parse(R"({"pi": 3.14, "numbers": [1, 2, 3]})"); + + std::cout << doc.root().is_object() << '\n'; + std::cout << doc.root().materialize().dump() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/json_document.output b/docs/mkdocs/docs/examples/json_document.output new file mode 100644 index 000000000..48225d164 --- /dev/null +++ b/docs/mkdocs/docs/examples/json_document.output @@ -0,0 +1,2 @@ +true +{"numbers":[1,2,3],"pi":3.14} diff --git a/docs/mkdocs/docs/examples/json_view.cpp b/docs/mkdocs/docs/examples/json_view.cpp new file mode 100644 index 000000000..64f8ec575 --- /dev/null +++ b/docs/mkdocs/docs/examples/json_view.cpp @@ -0,0 +1,14 @@ +#include +#include + +int main() +{ + std::cout << std::boolalpha; + + // json_view is basic_json_view: a read-only handle + // returned by json_document::root() + nlohmann::json_document doc = nlohmann::json_document::parse("[1, 2, 3]"); + nlohmann::json_view v = doc.root(); + + std::cout << v.is_array() << ' ' << v.size() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/json_view.output b/docs/mkdocs/docs/examples/json_view.output new file mode 100644 index 000000000..0a43dc098 --- /dev/null +++ b/docs/mkdocs/docs/examples/json_view.output @@ -0,0 +1 @@ +true 3 diff --git a/docs/mkdocs/docs/examples/json_view_ownership.cpp b/docs/mkdocs/docs/examples/json_view_ownership.cpp new file mode 100644 index 000000000..fa0e8262a --- /dev/null +++ b/docs/mkdocs/docs/examples/json_view_ownership.cpp @@ -0,0 +1,33 @@ +#include +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // BORROWED: doc only points into `text`; `text` must outlive `doc` + std::string text = R"({"a": 1})"; + json_document doc = json_document::parse(text); + std::cout << doc.owns_source() << '\n'; // false + + // a view is valid as long as the document is alive, has not been + // re-parsed (read()) or shrunk (shrink_to_fit()), and -- if borrowed -- + // the source text is alive + nlohmann::json_view v = doc.root(); + std::cout << v.is_object() << '\n'; + + // re-parsing the SAME document invalidates views taken before the call; + // `v` above must not be used after this line + doc.read(R"([1, 2, 3])"); + v = doc.root(); // take a fresh view instead + std::cout << v.is_array() << '\n'; + + // moving the document does not invalidate views: the node index is + // heap-allocated and does not move with the document object + json_document moved = std::move(doc); + std::cout << v.is_array() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/json_view_ownership.output b/docs/mkdocs/docs/examples/json_view_ownership.output new file mode 100644 index 000000000..fc2b492b1 --- /dev/null +++ b/docs/mkdocs/docs/examples/json_view_ownership.output @@ -0,0 +1,4 @@ +false +true +true +true diff --git a/docs/mkdocs/docs/examples/ordered_json_document.cpp b/docs/mkdocs/docs/examples/ordered_json_document.cpp new file mode 100644 index 000000000..2cae413bf --- /dev/null +++ b/docs/mkdocs/docs/examples/ordered_json_document.cpp @@ -0,0 +1,13 @@ +#include +#include + +using ordered_json_document = nlohmann::ordered_json_document; + +int main() +{ + // ordered_json_document is basic_json_document: + // materialize() preserves the insertion (source) order of object keys, + // instead of sorting them like json_document does + ordered_json_document doc = ordered_json_document::parse(R"({"z": 1, "a": 2, "m": 3})"); + std::cout << doc.root().materialize().dump() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/ordered_json_document.output b/docs/mkdocs/docs/examples/ordered_json_document.output new file mode 100644 index 000000000..4d441f752 --- /dev/null +++ b/docs/mkdocs/docs/examples/ordered_json_document.output @@ -0,0 +1 @@ +{"z":1,"a":2,"m":3} diff --git a/docs/mkdocs/docs/examples/ordered_json_view.cpp b/docs/mkdocs/docs/examples/ordered_json_view.cpp new file mode 100644 index 000000000..5fdf09a82 --- /dev/null +++ b/docs/mkdocs/docs/examples/ordered_json_view.cpp @@ -0,0 +1,10 @@ +#include +#include + +int main() +{ + nlohmann::ordered_json_document doc = nlohmann::ordered_json_document::parse(R"({"z": 1, "a": 2})"); + nlohmann::ordered_json_view v = doc.root(); + + std::cout << std::boolalpha << v.is_object() << ' ' << v.size() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/ordered_json_view.output b/docs/mkdocs/docs/examples/ordered_json_view.output new file mode 100644 index 000000000..fc59b1f20 --- /dev/null +++ b/docs/mkdocs/docs/examples/ordered_json_view.output @@ -0,0 +1 @@ +true 2 diff --git a/docs/mkdocs/docs/features/index.md b/docs/mkdocs/docs/features/index.md index aaf1253a7..d995c97d8 100644 --- a/docs/mkdocs/docs/features/index.md +++ b/docs/mkdocs/docs/features/index.md @@ -12,6 +12,8 @@ C++ types, and finally serialize it again. [JSON Lines](parsing/json_lines.md), [callbacks](parsing/parser_callbacks.md), the [SAX interface](parsing/sax_interface.md), [error handling](parsing/parse_exceptions.md), and [parsing untrusted input](parsing/untrusted_input.md). +- [Zero-copy JSON views](json_view.md) — read a JSON text through a flat index instead of building a `json` tree; + strings and numbers stay in the input and are only decoded when needed. - [Comments](comments.md) and [trailing commas](trailing_commas.md) — opt-in relaxations of the JSON grammar. ## Accessing and modifying values diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md new file mode 100644 index 000000000..6b66c42ca --- /dev/null +++ b/docs/mkdocs/docs/features/json_view.md @@ -0,0 +1,134 @@ +# Zero-copy JSON views + +`#!cpp ` adds a read-only, non-owning way to look at a parsed JSON text, as an alternative to +building a [`basic_json`](../api/basic_json/index.md) tree with [`parse()`](../api/basic_json/parse.md). + +## The problem + +[`basic_json::parse()`](../api/basic_json/parse.md) builds a tree of `basic_json` values: one allocation for every +array and object, and every string copied into its own `std::string`. That is the right trade-off when the program +goes on to read and write the value freely, but it does more work than necessary when only a small part of a large +JSON text is actually needed, or when the same text is parsed over and over (many small messages, for instance) and +most of the resulting tree is thrown away almost immediately. + +## The idea + +[`basic_json_document::parse()`](../api/basic_json_document/parse.md) parses the same JSON grammar, with the same +options, but instead of a tree it builds a flat index of the values it found: one +[16-byte entry](../home/architecture.md#node-index-of-json-views) per value (and one per object key), in document +order. Strings and numbers are not copied out of the input; they stay in the source text, and are only decoded when +actually needed (for a string, only if it contains escape sequences, into one shared buffer owned by the document). + +[`basic_json_view`](../api/basic_json_view/index.md) is a small, trivially copyable handle (two pointers) into that +index. It gives you the read-only, type-inspection part of the `basic_json` interface -- +[`type()`](../api/basic_json_view/type.md) and the `is_*()` predicates, +[`size()`](../api/basic_json_view/size.md)/[`empty()`](../api/basic_json_view/empty.md) -- without ever allocating a +`basic_json` value. When you do need an actual `basic_json` value for a subtree, +[`materialize()`](../api/basic_json_view/materialize.md) builds exactly the one +[`parse()`](../api/basic_json/parse.md) would have produced for it. + +## How to use it + +Include `` in addition to (or instead of) ``. Parse into a +[`json_document`](../api/json_document.md), inspect its [`root()`](../api/basic_json_document/root.md), and +`materialize()` when you need a real value: + +??? example "Example: parse a document, inspect its root, and materialize it" + + ```cpp + --8<-- "examples/json_document.cpp" + ``` + + Output: + + ```json + --8<-- "examples/json_document.output" + ``` + +[`ordered_json_document`](../api/ordered_json_document.md) is the equivalent for +[`ordered_json`](../api/ordered_json.md), just as [`ordered_json`](../api/ordered_json.md) is to +[`json`](../api/json.md). + +## Ownership and lifetime + +A document either **borrows** the text it was parsed from, or **owns** its own copy of it; call +[`owns_source()`](../api/basic_json_document/owns_source.md) to find out which happened. +[`parse()`](../api/basic_json_document/parse.md) decides this from the value category and type of its argument (an +lvalue `#!cpp std::string` is borrowed; an rvalue `#!cpp std::string` is moved in, owned without a copy; a stream is +read into an owned buffer; and so on -- see [`parse`'s Notes](../api/basic_json_document/parse.md#notes) for the full +table). [`parse_copy()`](../api/basic_json_document/parse_copy.md) always owns a copy, regardless of the input. + +!!! warning "A borrowed document depends on your buffer" + + If a document borrows its text, that text **must outlive the document** (and every view taken from it). Reading + or writing through a view after the underlying buffer is gone is undefined behavior, exactly as it would be for + a dangling `#!cpp std::string_view`. + +A view is valid only while all of the following hold: + +- the document is alive, +- the document has not been re-parsed since the view was taken (with [`read()`](../api/basic_json_document/read.md) + or [`parse()`](../api/basic_json_document/parse.md) into it), and has not had + [`shrink_to_fit()`](../api/basic_json_document/shrink_to_fit.md) called on it since, and +- if the document borrows its source text, that text is still alive. + +Moving the document itself is fine and does **not** invalidate its views: the index is a separate heap allocation +that keeps its address across the move. Take a fresh view from [`root()`](../api/basic_json_document/root.md) +whenever any of the other conditions above was not met. + +??? example "Example: borrowed and owned documents, and when views become invalid" + + ```cpp + --8<-- "examples/json_view_ownership.cpp" + ``` + + Output: + + ```json + --8<-- "examples/json_view_ownership.output" + ``` + +## What is the same as `parse()` + +- **Accept/reject.** [`accept()`](../api/basic_json_document/accept.md) and + [`parse()`](../api/basic_json_document/parse.md) accept and reject exactly the same inputs as + [`basic_json::accept()`](../api/basic_json/accept.md)/[`basic_json::parse()`](../api/basic_json/parse.md), with the + same `ignore_comments` and `ignore_trailing_commas` options. +- **Errors.** A failing parse throws the same exception -- the same id, message, and position -- because on failure + the library's own parser is run on the same bytes to produce the diagnostic. + `#!cpp allow_exceptions == false` gives a [discarded](../api/basic_json_document/is_discarded.md) document instead + of throwing, just as it gives a discarded value for `#!cpp basic_json::parse()`. +- **Number classification.** An integer literal that does not fit into the 64-bit integer type becomes a + floating-point value, exactly as it does for `#!cpp basic_json::parse()`. +- **Macros.** [`JSON_STRICT_NUL_HANDLING`](../api/macros/json_strict_nul_handling.md) and + [`JSON_NOEXCEPTION`](../api/macros/json_noexception.md)/[`JSON_THROW_USER`](../api/macros/json_throw_user.md) + behave the same way they do for ``. + +## What is different + +- **Only 64-bit integers.** `basic_json_document` requires `BasicJsonType::number_integer_t` and + `number_unsigned_t` to both be 64 bits wide; this is a compile-time `#!cpp static_assert`. +- **A 4 GiB input limit.** An input of 4 GiB or more throws + [`out_of_range.416`](../home/exceptions.md#jsonexceptionout_of_range416), a limit + `#!cpp basic_json::parse()` does not have. +- **A stream is always read to its end.** There is no partial/streaming read of an `#!cpp std::istream`. +- **No source positions on `materialize()`.** Even with + [`JSON_DIAGNOSTIC_POSITIONS`](../api/macros/json_diagnostic_positions.md) enabled, + [`materialize()`](../api/basic_json_view/materialize.md) does not set them: there is no lexer run during the + replay to record them. +- **Element access, iteration, `get()`, JSON Pointer, `dump()`, and comparison are not (yet) provided** by + `basic_json_view`. For now, [`materialize()`](../api/basic_json_view/materialize.md) is the way to get a value you + can do those things with. + +## Choosing between `json`, `ordered_json`, the SAX interface, and `json_view` + +| | [`json`](../api/json.md) / [`ordered_json`](../api/ordered_json.md) | [SAX interface](parsing/sax_interface.md) | [`json_document`](../api/json_document.md) / [`json_view`](../api/json_view.md) | +|---|---|---|---| +| **Ownership** | owns every value | owns nothing; you decide what to keep, in your handler | borrows or owns the *text*; the index is always owned by the document | +| **Mutability** | freely mutable | not applicable (a one-shot event stream) | read-only | +| **What you get** | a full tree you can read, write, and keep as long as you like | a sequence of callbacks; whatever your handler builds from them | a flat index plus, on demand, [`materialize()`](../api/basic_json_view/materialize.md)d `json`/`ordered_json` values for the parts you actually use | +| **Typical use** | general-purpose JSON handling: config, request/response bodies you build or modify, anything you hold onto | validating or projecting a text into your own data structure without ever holding the whole thing as JSON | large or high-volume input where you only need part of it, or need it repeatedly, and can keep the source text (or a copy) alive for as long as the document lives | + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/features/modules.md b/docs/mkdocs/docs/features/modules.md index 32e9ba7fd..28b538a2e 100644 --- a/docs/mkdocs/docs/features/modules.md +++ b/docs/mkdocs/docs/features/modules.md @@ -32,9 +32,15 @@ Only the following symbols are exported from `nlohmann.json`: - `nlohmann::adl_serializer` - `nlohmann::basic_json` +- `nlohmann::basic_json_document` +- `nlohmann::basic_json_view` - `nlohmann::json` +- `nlohmann::json_document` - `nlohmann::json_pointer` +- `nlohmann::json_view` - `nlohmann::ordered_json` +- `nlohmann::ordered_json_document` +- `nlohmann::ordered_json_view` - `nlohmann::ordered_map` - `nlohmann::to_string` - `nlohmann::literals::json_literals::operator""_json` diff --git a/docs/mkdocs/docs/features/performance.md b/docs/mkdocs/docs/features/performance.md index d6e9f70fd..87ccff787 100644 --- a/docs/mkdocs/docs/features/performance.md +++ b/docs/mkdocs/docs/features/performance.md @@ -55,7 +55,7 @@ always use the scalar path regardless of this macro. Parsing always produces SAX events internally; [`parse`](../api/basic_json/parse.md) simply feeds them to a consumer that builds a complete `basic_json` value tree (a DOM) in memory. For documents too large to comfortably hold as -a DOM, two alternatives avoid building it: +a DOM, three alternatives avoid building it: - Implement the [SAX interface](parsing/sax_interface.md) directly and pass it to [`sax_parse`](../api/basic_json/sax_parse.md); only the parts of the input you choose to keep ever become @@ -64,6 +64,12 @@ a DOM, two alternatives avoid building it: discard finished elements as soon as they are handled, so memory usage stays bounded by one element (plus the unparsed remainder of the input) instead of the whole document -- see the [recipe for streaming a large homogeneous array](parsing/parser_callbacks.md#recipe-streaming-a-large-homogeneous-array). +- Parse into a [`json_document`](json_view.md) (`#!cpp `) instead of a `basic_json`. It keeps + the input text and builds a flat index of 16 bytes per value; strings and numbers are not copied, but read from the + text when needed. Read-only [views](../api/basic_json_view/index.md) give the familiar element access, and only the + parts you [`materialize()`](../api/basic_json_view/materialize.md) become `basic_json` values. A document that + borrows the text instead of owning a copy needs the text to outlive it; see + [choosing between `json`, the SAX interface, and `json_view`](json_view.md#choosing-between-json-ordered_json-the-sax-interface-and-json_view). If the data is naturally record-oriented, consider [JSON Lines](parsing/json_lines.md) instead of one large JSON document: reading and parsing it line by line with `#!cpp std::getline` means only one line's value is ever in memory @@ -209,6 +215,7 @@ those headers are then never processed by the compiler at all. - [Architecture](../home/architecture.md) - how input adapters, the lexer, and the serializer fit together - [Parsing](parsing/index.md) - the available parsing functions and inputs - [SAX interface](parsing/sax_interface.md) - parse without building a DOM +- [Zero-copy JSON views](json_view.md) - parse into a flat index of the text and read it without building a DOM - [Binary formats](binary_formats/index.md) - compact alternatives to JSON text - [Object Order](object_order.md) - `json` vs. `ordered_json` and other `ObjectType` choices - [Template Parameter Requirements](types/template_parameters.md) - custom container and allocator types diff --git a/docs/mkdocs/docs/home/architecture.md b/docs/mkdocs/docs/home/architecture.md index 6c0892f11..457f0d536 100644 --- a/docs/mkdocs/docs/home/architecture.md +++ b/docs/mkdocs/docs/home/architecture.md @@ -51,6 +51,11 @@ The public headers are in [`include/nlohmann`](https://github.com/nlohmann/json/ [`adl_serializer`](../api/adl_serializer/index.md), [`byte_container_with_subtype`](../api/byte_container_with_subtype/index.md), and [`ordered_map`](../api/ordered_map.md). +- [`json_view.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/json_view.hpp) is a separate, + optional header that defines [`basic_json_document`](../api/basic_json_document/index.md) and + [`basic_json_view`](../api/basic_json_view/index.md), a flat-index, read-only, non-owning way to look at a parsed + JSON text; see [Zero-copy JSON views](../features/json_view.md). It builds on `json.hpp` internals (it requires the + same library version) and has its own `detail/view/` subdirectory. Everything else lives in [`detail/`](https://github.com/nlohmann/json/tree/develop/include/nlohmann/detail) and namespace `nlohmann::detail`, which is not part of the public API. Paths below are relative to `include/nlohmann`. @@ -74,7 +79,9 @@ below are relative to `include/nlohmann`. | Macros | [`detail/macro_scope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_scope.hpp), [`detail/macro_unscope.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/macro_unscope.hpp), [`detail/abi_macros.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/abi_macros.hpp) | The single-header version [`single_include/nlohmann/json.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json.hpp) -is generated from these files with `make amalgamate` and must not be edited by hand. +is generated from these files with `make amalgamate` and must not be edited by hand. The same command also generates +[`single_include/nlohmann/json_view.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json_view.hpp) +from `json_view.hpp` and `detail/view/`. ## Template parameters @@ -164,6 +171,74 @@ pointer to them. This keeps a `basic_json` value small: one pointer-sized union maintains the invariant that the pointer matching `m_type` is never null; `assert_invariant()` checks it with [runtime assertions](../features/assertions.md). +## Node index of JSON views + +A [`basic_json_document`](../api/basic_json_document/index.md) (see [Zero-copy JSON views](../features/json_view.md)) +does not build a tree of values. Its parser +([`detail/view/builder.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/builder.hpp)) +writes a flat array of 16-byte nodes, one per value and one per object key, in document order. A +[`basic_json_view`](../api/basic_json_view/index.md) is a pointer to the document and a pointer to one node. The layout +is `struct node` in +[`detail/view/node.hpp`](https://github.com/nlohmann/json/blob/develop/include/nlohmann/detail/view/node.hpp) +(the numbers are bit offsets, 32 bits per row): + +```mermaid +packet-beta + 0-7: "kind" + 8-15: "flags" + 16-31: "extra" + 32-63: "off" + 64-95: "len" + 96-127: "next" +``` + +| Bytes | Field | Type | Contents | +|-------|---------|------------|-------------------------------------------------------------------------------------------------------------------------------| +| 0 | `kind` | `uint8_t` | the type, numbered as [`value_t`](../api/basic_json/value_t.md): 0 null, 1 object, 2 array, 3 string, 4 boolean, 5 signed integer, 6 unsigned integer, 7 float | +| 1 | `flags` | `uint8_t` | bits 0-1: where a string's bytes are (0: the source text, 1: the buffer of decoded strings, for strings with escapes); bit 2: the value of a boolean | +| 2-3 | `extra` | `uint16_t` | numbers: the number of integer digits (low byte) and fraction digits (high byte), 255 for more; otherwise 0 | +| 4-7 | `off` | `uint32_t` | where the value starts: the first byte after a string's opening quote (or its position in the buffer of decoded strings), the first byte of a number or literal, the bracket of an array or object | +| 8-11 | `len` | `uint32_t` | strings: the length after decoding; floats and literals: the length of the token; arrays and objects: the number of elements | +| 12-15 | `next` | `uint32_t` | arrays and objects: the number of nodes of the subtree, including the node itself | + +- **Integers** keep their converted value in bytes 8-15 instead of `len` and `next`; the length of their token follows + from the number of digits in `extra` (and the sign). Non-negative integers are unsigned integers, as with + [`parse`](../api/basic_json/parse.md). +- **Floats** keep only their token. The digit layout in `extra` lets the conversion read the digits without scanning the + token again, and only when the value is read. +- **Object members** are the node of the key (a string) followed by the nodes of the value. +- **Navigation** needs no pointers: the elements of an array or object follow its node, and the node after a value's + subtree is `next` nodes further for an array or object, and the next node otherwise (`document_data::after`). Views + step from element to element this way and skip whole subtrees in constant time. +- **Offsets** are 32 bits wide, so a document is limited to 4 GiB (`out_of_range.416`). + +For example, `#!json {"a": [1, 2.5]}` becomes five nodes. Each node's elements follow it, and `next` leads from an +array or object past its subtree: + +```mermaid +flowchart LR + n0["0: object
len 1, next 5"] + n1["1: key a"] + n2["2: array
len 2, next 3"] + n3["3: unsigned integer 1"] + n4["4: float 2.5"] + e(["end"]) + n0 --> n1 --> n2 --> n3 --> n4 --> e + n0 -. next .-> e + n2 -. next .-> e +``` + +| Node | `kind` | `extra` | `off` | `len` | `next` | +|------|----------------------|---------|-------|-------|--------| +| 0 | 1 (object) | 0 | 0 | 1 | 5 | +| 1 | 3 (string) | 0 | 2 | 1 | 0 | +| 2 | 2 (array) | 0 | 6 | 2 | 3 | +| 3 | 6 (unsigned integer) | 0x0001 | 7 | - | - | +| 4 | 7 (float) | 0x0101 | 10 | 3 | 0 | + +All `flags` are 0. The integer's bytes 8-15 hold its value, 1; its `extra` says it has one digit. The float's `extra` +says it has one integer and one fraction digit, and its `len` is that of the token `2.5`. + ## Input adapters Input is read via **input adapters** that abstract a source. Every input adapter provides this interface: diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 9cb5c8f62..15673d5c7 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -1043,6 +1043,23 @@ MessagePack's ext type and BSON's binary subtype are each stored in a single byt This exception was added in version 3.13.0. Before that, subtypes above 255 were silently truncated modulo 256 instead of raising an error. +### json.exception.out_of_range.416 + +[`basic_json_document::parse()`](../api/basic_json_document/parse.md) and the other parsing functions of +[`basic_json_document`](../api/basic_json_document/index.md) index a value's position in the source text in 32 bits, +so they do not support an input of 4 GiB or more. + +!!! failure "Example message" + + ``` + [json.exception.out_of_range.416] input of 4 GiB or more is not supported by json_document + ``` + +!!! note + + This exception was added in version 3.13.0, together with [``](../features/json_view.md). + [`basic_json::parse()`](../api/basic_json/parse.md) has no such limit. + ## Further exceptions This exception is thrown in case of errors that cannot be classified with the diff --git a/docs/mkdocs/docs/home/license.md b/docs/mkdocs/docs/home/license.md index d3ce12e28..3327ad791 100644 --- a/docs/mkdocs/docs/home/license.md +++ b/docs/mkdocs/docs/home/license.md @@ -23,3 +23,5 @@ The class contains a port of the shortest double-to-decimal conversion of [Żmij The class contains a copy of [Hedley](https://nemequ.github.io/hedley/) from Evan Nemerson which is licensed as [CC0-1.0](https://creativecommons.org/publicdomain/zero/1.0/). The class contains an adapted version of the Eisel-Lemire algorithm, its table of powers of five, and its digit comparison for long numbers from [fast_float](https://github.com/fastfloat/fast_float) by Daniel Lemire and contributors, which is available under the [MIT License](https://opensource.org/licenses/MIT) (used here), the Apache 2.0 License, and the Boost Software License. Copyright © 2021 The fast_float authors + +The view's parser (``) contains techniques and code adapted from [yyjson](https://github.com/ibireme/yyjson) by YaoYuan, which is licensed under the [MIT License](https://opensource.org/licenses/MIT) (see above): table-driven decoding of `\u` escapes and fixed-offset unrolled checks. diff --git a/docs/mkdocs/docs/integration/index.md b/docs/mkdocs/docs/integration/index.md index 846c15fdc..f022ec6ed 100644 --- a/docs/mkdocs/docs/integration/index.md +++ b/docs/mkdocs/docs/integration/index.md @@ -49,3 +49,8 @@ for forward declarations (see [Compile times](compile_times.md)), and file [`single_include/nlohmann/json_literals.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json_literals.hpp) for the user-defined string literals if you define [`JSON_NO_AUTOMATIC_UDLS`](../api/macros/json_no_automatic_udls.md). + +For the read-only, non-owning [`basic_json_document`](../api/basic_json_document/index.md)/[`basic_json_view`](../api/basic_json_view/index.md) +types, additionally include +[`single_include/nlohmann/json_view.hpp`](https://github.com/nlohmann/json/blob/develop/single_include/nlohmann/json_view.hpp); +see [Zero-copy JSON views](../features/json_view.md). diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 5b23366bb..6e6a2c4cd 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -102,6 +102,7 @@ nav: - features/types/index.md - features/types/number_handling.md - features/types/template_parameters.md + - features/json_view.md - Integration: - integration/index.md - integration/migration_guide.md @@ -232,6 +233,42 @@ nav: - 'update': api/basic_json/update.md - 'value': api/basic_json/value.md - 'value_t': api/basic_json/value_t.md + - basic_json_document: + - 'Overview': api/basic_json_document/index.md + - '(Constructor)': api/basic_json_document/basic_json_document.md + - 'accept': api/basic_json_document/accept.md + - 'is_discarded': api/basic_json_document/is_discarded.md + - 'memory_usage': api/basic_json_document/memory_usage.md + - 'node_count': api/basic_json_document/node_count.md + - 'owns_source': api/basic_json_document/owns_source.md + - 'parse': api/basic_json_document/parse.md + - 'parse_copy': api/basic_json_document/parse_copy.md + - 'read': api/basic_json_document/read.md + - 'root': api/basic_json_document/root.md + - 'shrink_to_fit': api/basic_json_document/shrink_to_fit.md + - 'source': api/basic_json_document/source.md + - basic_json_view: + - 'Overview': api/basic_json_view/index.md + - '(Constructor)': api/basic_json_view/basic_json_view.md + - 'empty': api/basic_json_view/empty.md + - 'is_array': api/basic_json_view/is_array.md + - 'is_binary': api/basic_json_view/is_binary.md + - 'is_boolean': api/basic_json_view/is_boolean.md + - 'is_discarded': api/basic_json_view/is_discarded.md + - 'is_null': api/basic_json_view/is_null.md + - 'is_number': api/basic_json_view/is_number.md + - 'is_number_float': api/basic_json_view/is_number_float.md + - 'is_number_integer': api/basic_json_view/is_number_integer.md + - 'is_number_unsigned': api/basic_json_view/is_number_unsigned.md + - 'is_object': api/basic_json_view/is_object.md + - 'is_primitive': api/basic_json_view/is_primitive.md + - 'is_string': api/basic_json_view/is_string.md + - 'is_structured': api/basic_json_view/is_structured.md + - 'materialize': api/basic_json_view/materialize.md + - 'operator bool': api/basic_json_view/operator_bool.md + - 'size': api/basic_json_view/size.md + - 'source_offset': api/basic_json_view/source_offset.md + - 'type': api/basic_json_view/type.md - byte_container_with_subtype: - 'Overview': api/byte_container_with_subtype/index.md - '(constructor)': api/byte_container_with_subtype/byte_container_with_subtype.md @@ -246,6 +283,7 @@ nav: - 'from_json': api/adl_serializer/from_json.md - 'to_json': api/adl_serializer/to_json.md - 'json': api/json.md + - 'json_document': api/json_document.md - json_pointer: - 'Overview': api/json_pointer/index.md - '(Constructor)': api/json_pointer/json_pointer.md @@ -280,11 +318,14 @@ nav: - 'start_array': api/json_sax/start_array.md - 'start_object': api/json_sax/start_object.md - 'string': api/json_sax/string.md + - 'json_view': api/json_view.md - 'operator<<(basic_json), operator<<(json_pointer)': api/operator_ltlt.md - 'operator>>(basic_json)': api/operator_gtgt.md - 'operator""_json': api/operator_literal_json.md - 'operator""_json_pointer': api/operator_literal_json_pointer.md - 'ordered_json': api/ordered_json.md + - 'ordered_json_document': api/ordered_json_document.md + - 'ordered_json_view': api/ordered_json_view.md - 'ordered_map': api/ordered_map.md - macros: - 'Overview': api/macros/index.md @@ -472,6 +513,8 @@ plugins: API Documentation: - api/*.md - api/basic_json/*.md + - api/basic_json_document/*.md + - api/basic_json_view/*.md - api/adl_serializer/*.md - api/byte_container_with_subtype/*.md - api/json_pointer/*.md diff --git a/include/nlohmann/detail/abi_config.hpp b/include/nlohmann/detail/abi_config.hpp new file mode 100644 index 000000000..0e99c81ad --- /dev/null +++ b/include/nlohmann/detail/abi_config.hpp @@ -0,0 +1,34 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief the configuration macros that change the library's behavior + +json.hpp undefines these macros at its end (see macro_unscope.hpp), so code +that builds on the library after it (json_view.hpp) reads them here. Like the +macros, they are part of the ABI namespace, so they always match the +basic_json they are used with. +*/ +struct abi_config +{ + /// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input + static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0; + /// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0; +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/builder.hpp b/include/nlohmann/detail/view/builder.hpp new file mode 100644 index 000000000..1cd01fef6 --- /dev/null +++ b/include/nlohmann/detail/view/builder.hpp @@ -0,0 +1,1035 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2020 YaoYuan +// SPDX-License-Identifier: MIT + +#pragma once + +#include // find, find_if, max +#include // array +#include // size_t, ptrdiff_t +#include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t +#include // memcmp, memcpy +#include // numeric_limits +#include // string +#include // vector + +#include +#include +#include +#include +#include + +// The view's parser: one pass over the input that emits the node index (see +// node.hpp). The table-driven decoding of \u escapes and the fast paths for +// ": " and indentation follow yyjson (https://github.com/ibireme/yyjson, MIT +// license). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +enum class error_code : std::uint8_t +{ + none, + empty_input, + unexpected_value, + invalid_literal, + expected_key, + expected_colon, + expected_array_end, + expected_object_end, + trailing_characters, + number_after_minus, + number_after_dot, + number_after_exponent, + number_overflow, + string_missing_quote, + string_control_character, + string_utf8, + string_escape, + string_unicode_hex, + string_surrogate_high, + string_surrogate_low, + comment_start, + comment_unterminated, + input_too_large, +}; + +struct parse_failure +{ + error_code code = error_code::none; + std::size_t offset = 0; ///< byte offset of the offending character +}; + +/// FloatType: the number_float_t of the document, whose overflow parse() rejects +template +class builder +{ + public: + builder(document_data& d, const char* src, std::size_t size) noexcept + : doc(d) + , b(reinterpret_cast(src)) + , e(b + size) + {} + + /// returns false and fills `failure` on error + bool run() + { + cursor c(*this); + return c.run(); + } + + builder(const builder&) = delete; + builder& operator=(const builder&) = delete; + builder(builder&&) = delete; + builder& operator=(builder&&) = delete; + ~builder() = default; + + /// where and why the parse failed (after run() returned false) + const parse_failure& failure() const noexcept + { + return m_failure; + } + + private: + struct frame + { + std::uint32_t idx; + std::uint32_t count; + bool is_object; + }; + + document_data& doc; + const unsigned char* const b; + const unsigned char* const e; + parse_failure m_failure{}; + + // the open array/object is in the cursor; enclosing ones on a stack that is + // inline for the first 64 levels + frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open + std::vector deep{}; + + NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept + { + m_failure.code = c; + m_failure.offset = static_cast(at - b); + doc.tape_size = 0; + return false; + } + + /// a failure (recorded by fail) as a comment() result + const unsigned char* fail_at(error_code c, const unsigned char* at) noexcept + { + fail(c, at); + return nullptr; + } + + /// a decoded string: p after its closing quote (nullptr: an error), and + /// its bytes in the arena + struct decoded + { + const unsigned char* p; + std::size_t start; + std::size_t len; + }; + + decoded failed(error_code c, const unsigned char* at) noexcept + { + fail(c, at); + return decoded{nullptr, 0, 0}; + } + + /// the comment at p (*p == '/'): the position after it, or nullptr on error + NLOHMANN_VIEW_NOINLINE const unsigned char* comment(const unsigned char* p) + { + if (e - p < 2) + { + ++p; + return fail_at(error_code::comment_start, p); + } + if (p[1] == '/') + { + p += 2; + // (as in parse(), a null byte is the end of the input, so it is + // left for the caller to see) + while (p != e && *p != '\n' && *p != '\r' && !(NulIsEnd && *p == 0)) + { + ++p; + } + return p; + } + if (p[1] == '*') + { + p += 2; + for (;;) + { + if (p == e || (NulIsEnd && *p == 0)) + { + return fail_at(error_code::comment_unterminated, p); + } + if (*p == '*' && p + 1 != e && p[1] == '/') + { + p += 2; + return p; + } + ++p; + } + } + ++p; + return fail_at(error_code::comment_start, p); + } + + /// the index is full (n nodes, parsed up to at): extrapolate the node + /// count from the nodes per input byte so far (with headroom, and at least + /// 1.5 times as many), so that dense inputs regrow once instead of + /// doubling repeatedly; returns the new node array + NLOHMANN_VIEW_NOINLINE node* grow(std::size_t n, const unsigned char* at) + { + const std::uint64_t done = static_cast(at - b) + 1; + const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; + const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t + doc.tape_size = n; + doc.reserve((std::max)(static_cast(grown), n + (n / 2) + 64)); + return doc.tape; + } + + /// four hex digits at p as a code unit (p moves past them), or -1 (p at + /// the first bad digit); one table lookup per digit and a single check, + /// the four hex digits of a unicode escape (the library's table, after + /// yyjson's read_hex_u16), or -1 + NLOHMANN_VIEW_ALWAYS_INLINE int hex4(const unsigned char*& p) noexcept + { + if (NLOHMANN_VIEW_LIKELY(e - p >= 4)) + { + const int cp = hex_codepoint(p); + if (NLOHMANN_VIEW_LIKELY(cp >= 0)) + { + p += 4; + return cp; + } + } + p = hex4_error(p); + return -1; + } + + /// hex4() failed: the first bad digit (none: p) + NLOHMANN_VIEW_NOINLINE const unsigned char* hex4_error(const unsigned char* p) noexcept + { + if (e - p >= 4) + { + while (is_hex(*p)) + { + ++p; + } + } + return p; + } + + /// escapes present (or an error) in the string at s, scanned up to p: + /// decode into the arena + NLOHMANN_VIEW_NOINLINE decoded slow_string(const unsigned char* s, const unsigned char* p) + { + // single-character escapes; 0: invalid (and 'u', handled separately) + static const std::array simple_escape = + { + { + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, '"', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '/', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '\\', 0, 0, 0, + 0, 0, '\b', 0, 0, 0, '\f', 0, 0, 0, 0, 0, 0, 0, '\n', 0, 0, 0, '\r', 0, '\t', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0 + } + }; + const std::size_t start = arena_used(); + arena_run(s, static_cast(p - s)); + for (;;) + { + if (p == e) + { + return failed(error_code::string_missing_quote, p); + } + const unsigned char c = *p; + if (c == '"') + { + ++p; + return decoded{p, start, arena_used() - start}; + } + if (c != '\\') + { + // (a NUL before the end of the input is a control character, as + // for json::parse, also where a NUL ends the input between values) + return failed(c < 0x20 ? error_code::string_control_character : error_code::string_utf8, p); + } + ++p; + if (p == e) + { + return failed(error_code::string_missing_quote, p); + } + const unsigned char d = *p++; + arena_ensure(4); + if (d == 'u') + { + int cp = hex4(p); + if (NLOHMANN_VIEW_UNLIKELY(cp < 0)) + { + return failed(error_code::string_unicode_hex, p); + } + if (NLOHMANN_VIEW_UNLIKELY((cp & 0xF800) == 0xD800)) // a surrogate + { + if (cp >= 0xDC00) + { + return failed(error_code::string_surrogate_low, p); + } + if (e - p < 2 || p[0] != '\\' || p[1] != 'u') + { + return failed(error_code::string_surrogate_high, p); + } + p += 2; + const int lo = hex4(p); + if (lo < 0) + { + return failed(error_code::string_unicode_hex, p); + } + if (lo < 0xDC00 || lo > 0xDFFF) + { + return failed(error_code::string_surrogate_high, p); + } + cp = 0x10000 + ((cp - 0xD800) << 10) + (lo - 0xDC00); + } + aw = put_utf8(aw, cp); + } + else if (NLOHMANN_VIEW_LIKELY(d < 128 && simple_escape[d] != 0)) + { + *aw++ = simple_escape[d]; + } + else + { + --p; + return failed(error_code::string_escape, p); + } + if (p != e && *p == '\\') + { + continue; // consecutive escapes ("\u00e4\u00f6"): no run in between + } + const unsigned char* const r = p; + p = scan_string_run(p, e); + arena_run(r, static_cast(p - r)); + } + } + + /// does the float token [s, p) overflow FloatType? (parse() rejects it) + NLOHMANN_VIEW_NOINLINE static bool float_overflows(const unsigned char* s, const unsigned char* p) + { + const auto* const first = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const auto* const last = reinterpret_cast(p); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const char* const dot = std::find(first, last, '.'); + const char* const exponent = std::find_if(first, last, [](char c) + { + return c == 'e' || c == 'E'; + }); + const auto v = convert_float(first, last, dot == last ? std::string::npos : static_cast(dot - first), + static_cast(exponent - first)); + return v > (std::numeric_limits::max)() || v < -(std::numeric_limits::max)(); + } + + /// does the magnitude digits [d, d + n) exceed the given limit (same length)? + static bool digits_exceed(const unsigned char* d, const char* limit, std::size_t n) noexcept + { + return std::memcmp(d, limit, n) > 0; + } + + static bool is_hex(unsigned char c) noexcept + { + return (c >= '0' && c <= '9') || (c >= 'A' && c <= 'F') || (c >= 'a' && c <= 'f'); + } + + // decode arena: the std::string in the document, written through a raw + // pointer (resized ahead in large steps; trimmed when parsing succeeds) + char* aw = nullptr; + char* aend = nullptr; + + NLOHMANN_VIEW_ALWAYS_INLINE std::size_t arena_used() const noexcept + { + return aw != nullptr ? static_cast(aw - doc.arena.data()) : 0; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void arena_ensure(std::size_t n) + { + if (NLOHMANN_VIEW_UNLIKELY(static_cast(aend - aw) < n)) + { + arena_grow(n); + } + } + + NLOHMANN_VIEW_NOINLINE void arena_grow(std::size_t n) + { + const std::size_t used = arena_used(); + doc.arena.resize((std::max)(doc.arena.size() * 2, used + n + 256)); + aw = &doc.arena[0] + used; // NOLINT(readability-container-data-pointer): data() is const before C++17 + aend = &doc.arena[0] + doc.arena.size(); // NOLINT(readability-container-data-pointer) + } + + /// append the run [r, r + n) to the arena; short runs as one fixed-size + /// 16-byte move when both sides have the room (no library call) + NLOHMANN_VIEW_ALWAYS_INLINE void arena_run(const unsigned char* r, std::size_t n) + { + arena_ensure(n + 16); + if (n <= 16 && e - r >= 16) + { + std::memcpy(aw, r, 16); + } + else + { + std::memcpy(aw, r, n); + } + aw += n; + } + + /// UTF-8 encoding of cp at w (room for 4 bytes) + static char* put_utf8(char* w, int cp) noexcept + { + if (cp < 0x80) + { + *w++ = static_cast(cp); + } + else if (cp < 0x800) + { + *w++ = static_cast(0xC0 | (cp >> 6)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + else if (cp < 0x10000) + { + *w++ = static_cast(0xE0 | (cp >> 12)); + *w++ = static_cast(0x80 | ((cp >> 6) & 0x3F)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + else + { + *w++ = static_cast(0xF0 | (cp >> 18)); + *w++ = static_cast(0x80 | ((cp >> 12) & 0x3F)); + *w++ = static_cast(0x80 | ((cp >> 6) & 0x3F)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + return w; + } + + /// a compile-time option as a runtime condition: testing the template + /// argument directly makes a condition like `TrailingCommas && c == ']'` + /// constant when the option is off, which MSVC reports as C4127 + static NLOHMANN_VIEW_ALWAYS_INLINE bool enabled(bool option) noexcept + { + return option; + } + + /// The parse state and the parser proper. The cursor is a local object of + /// run() whose address never escapes (everything it calls out of line is a + /// member of the builder and gets the positions it needs), so that the + /// compiler keeps the state in registers instead of reloading it from + /// memory after every node store and call. + struct cursor + { + explicit cursor(builder& owner) noexcept + : cold(owner) + , b(owner.b) + , p(owner.b) + , e(owner.e) + {} + + builder& cold; ///< out-of-line helpers and state that needs no registers + const unsigned char* const b; + const unsigned char* p; + const unsigned char* const e; + node* base = nullptr; + node* out = nullptr; + node* cap = nullptr; + + // the open array/object + std::uint32_t cur_idx = 0; + std::uint32_t cur_count = 0; + bool cur_is_object = false; + std::size_t depth = 0; + + NLOHMANN_VIEW_ALWAYS_INLINE bool run() + { + cold.doc.reserve(estimate_nodes(reinterpret_cast(b), static_cast(e - b))); + base = cold.doc.tape; + out = base; + cap = base + cold.doc.tape_cap; + + if (e - p >= 3 && p[0] == 0xEF && p[1] == 0xBB && p[2] == 0xBF) + { + p += 3; // byte order mark + } + if (!ws()) + { + return false; + } + if (p == e || (NulIsEnd && *p == 0)) + { + return fail(error_code::empty_input); + } + + // root value + switch (cur()) + { + case '{': + open(value_t::object); + ++p; + goto obj_first; + case '[': + open(value_t::array); + ++p; + goto arr_first; + default: + if (!scalar()) + { + return false; + } + goto root_done; + } + + // value dispatch, expanded once for array elements and once for member + // values: each jump has its own history (arrays tend to hold one kind of + // value), and the continuation needs no branch on the container kind +#define NLOHMANN_VIEW_VALUE(NEXT) \ + switch (cur()) \ + { \ + case '"': \ + if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \ + goto NEXT; \ + case '{': \ + open(value_t::object); \ + ++p; \ + goto obj_first; \ + case '[': \ + open(value_t::array); \ + ++p; \ + goto arr_first; \ + case '-': \ + if (NLOHMANN_VIEW_UNLIKELY(!number())) { return false; } \ + goto NEXT; \ + case '0': case '1': case '2': case '3': \ + case '4': case '5': case '6': case '7': case '8': case '9': \ + if (NLOHMANN_VIEW_UNLIKELY(!number())) { return false; } \ + goto NEXT; \ + case 't': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal("true", 4, value_t::boolean, node_flags::is_true))) { return false; } \ + goto NEXT; \ + case 'f': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal_false())) { return false; } \ + goto NEXT; \ + case 'n': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal("null", 4, value_t::null, 0))) { return false; } \ + goto NEXT; \ + default: \ + return fail(error_code::unexpected_value); \ + } + +arr_first: + if (!ws()) + { + return false; + } + if (cur() == ']') + { + ++p; + goto close_container; + } +value: + NLOHMANN_VIEW_VALUE(arr_next) +arr_next: + ++cur_count; + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ',')) + { + ++p; + if (!ws()) + { + return false; + } + if (enabled(TrailingCommas) && cur() == ']') + { + ++p; + goto close_container; + } + goto value; + } + if (cur() == ']') + { + ++p; + goto close_container; + } + return fail(error_code::expected_array_end); + +obj_first: + if (!ws()) + { + return false; + } + if (cur() == '}') + { + ++p; + goto close_container; + } +obj_key: + if (NLOHMANN_VIEW_UNLIKELY(cur() != '"')) + { + return fail(error_code::expected_key); + } + if (NLOHMANN_VIEW_UNLIKELY(!string())) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ':' && (Sentinel || e - p >= 2) && p[1] == ' ')) + { + p += 2; // ": " (pretty-printed input; a fast path of yyjson) + } + else + { + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_UNLIKELY(cur() != ':')) + { + return fail(error_code::expected_colon); + } + ++p; + } + if (!ws()) + { + return false; + } + NLOHMANN_VIEW_VALUE(obj_next) +obj_next: + ++cur_count; + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ',')) + { + ++p; + if (!ws()) + { + return false; + } + if (enabled(TrailingCommas) && cur() == '}') + { + ++p; + goto close_container; + } + goto obj_key; + } + if (cur() == '}') + { + ++p; + goto close_container; + } + return fail(error_code::expected_object_end); + +#undef NLOHMANN_VIEW_VALUE + +close_container: + close(); + if (NLOHMANN_VIEW_UNLIKELY(depth == 0)) + { + goto root_done; + } + if (cur_is_object) + { + goto obj_next; + } + goto arr_next; + +root_done: + if (!ws()) + { + return false; + } + if (p != e && !(NulIsEnd && *p == 0)) + { + return fail(error_code::trailing_characters); + } + cold.doc.tape_size = static_cast(out - base); + cold.doc.arena.resize(cold.arena_used()); + return true; + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool fail(error_code c) noexcept + { + return cold.fail(c, p); + } + + /// the current byte, or 0 at the end. With a NUL-terminated input + /// (Sentinel) the terminator is read instead of checking the bounds; a 0 + /// never matches a JSON token, so the error paths tell the end apart. + NLOHMANN_VIEW_ALWAYS_INLINE unsigned char cur() const noexcept + { + if (Sentinel) + { + return *p; + } + return p != e ? *p : 0; + } + + /// root scalar + NLOHMANN_VIEW_ALWAYS_INLINE bool scalar() + { + switch (cur()) + { + case '"': + return string(); + case 't': + return literal("true", 4, value_t::boolean, node_flags::is_true); + case 'f': + return literal_false(); + case 'n': + return literal("null", 4, value_t::null, 0); + case '-': + return number(); + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + return number(); + default: + return fail(error_code::unexpected_value); + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool literal_false() + { + return literal("false", 5, value_t::boolean, 0); + } + + /// skip whitespace (and comments); false on a malformed comment + NLOHMANN_VIEW_ALWAYS_INLINE bool ws() + { + const unsigned char c = cur(); + if (NLOHMANN_VIEW_LIKELY(c > ' ' && (!Comments || c != '/'))) + { + return true; // no whitespace: the common case in minified input + } + return ws_slow(); + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool ws_slow() + { + for (;;) + { + if (cur() == ' ' && (Sentinel || e - p >= 2) && p[1] > ' ' && (!Comments || p[1] != '/')) + { + ++p; // single space, e.g. after ':' or ',' + return true; + } + if (cur() == '\n' || cur() == '\r') + { + // (a branch, not an add of the comparison: p must not + // wait for the byte after the line break) + if (NLOHMANN_VIEW_UNLIKELY(cur() == '\r') && (Sentinel || e - p >= 2) && p[1] == '\n') + { + p += 2; + } + else + { + ++p; + } + // indentation: two spaces per step, fixed offsets (after yyjson) + while (e - p >= 32) + { +#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(load16(p + (std::ptrdiff_t{2} * (i))) == 0x2020)) {} else { p += std::ptrdiff_t{2} * (i); goto indent_done; } + NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) +#undef NLOHMANN_VIEW_STEP + p += 32; + } +indent_done: + ; + } + for (unsigned char c = cur(); c == ' ' || c == '\n' || c == '\r' || c == '\t'; c = cur()) + { + ++p; + } + if (enabled(Comments) && cur() == '/') + { + const unsigned char* const q = cold.comment(p); + if (q == nullptr) + { + return false; + } + p = q; + continue; + } + return true; + } + } + + /// append a node: (kind, flags, extra, off) and the second word (len, or + /// an integer's value); two stores on little-endian targets + NLOHMANN_VIEW_ALWAYS_INLINE node* emit(value_t k, std::uint8_t flags, std::uint16_t extra, std::size_t off, std::uint64_t second) + { + if (NLOHMANN_VIEW_UNLIKELY(out == cap)) + { + const auto n = static_cast(out - base); + base = cold.grow(n, p); + out = base + n; + cap = base + cold.doc.tape_cap; + } + node* n = out++; +#if NLOHMANN_VIEW_LITTLE_ENDIAN + const std::uint64_t first = static_cast(k) | (static_cast(flags) << 8) + | (static_cast(extra) << 16) | (static_cast(off) << 32); + std::memcpy(reinterpret_cast(n), &first, 8); + std::memcpy(reinterpret_cast(n) + 8, &second, 8); +#else + n->kind = static_cast(k); + n->flags = flags; + n->extra = extra; + n->off = static_cast(off); + set_integer_bits(*n, second); +#endif + return n; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void open(value_t k) + { + const auto idx = static_cast(emit(k, 0, 0, static_cast(p - b), 0) - base); + if (depth != 0) + { + const frame f = {cur_idx, cur_count, cur_is_object}; + if (NLOHMANN_VIEW_LIKELY(depth <= 64)) + { + cold.shallow[depth - 1] = f; + } + else + { + cold.deep.push_back(f); + } + } + ++depth; + cur_idx = idx; + cur_count = 0; + cur_is_object = k == value_t::object; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void close() + { + node& n = base[cur_idx]; + n.len = cur_count; + n.next = static_cast(out - base) - cur_idx; + if (--depth != 0) + { + frame f{}; + if (NLOHMANN_VIEW_LIKELY(depth <= 64)) + { + f = cold.shallow[depth - 1]; + } + else + { + f = cold.deep.back(); + cold.deep.pop_back(); + } + cur_idx = f.idx; + cur_count = f.count; + cur_is_object = f.is_object; + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool literal(const char* text, std::size_t n, value_t k, std::uint8_t flags) + { + if (NLOHMANN_VIEW_UNLIKELY(e - p < static_cast(n) || std::memcmp(p, text, n) != 0)) + { + return fail(error_code::invalid_literal); + } + emit(k, flags, 0, static_cast(p - b), n); + p += n; + return true; + } + + /// a number at p; the sign is known from the dispatch (so that p does + /// not have to wait for the first byte) + template + NLOHMANN_VIEW_ALWAYS_INLINE bool number() + { + const unsigned char* const s = p; + if (negative) + { + ++p; + } + const unsigned char* const int_start = p; + if (p != e && *p == '0') + { + ++p; + } + else if (NLOHMANN_VIEW_LIKELY(p != e && *p >= '1' && *p <= '9')) + { + p = skip_digits(p + 1, e); + } + else + { + return fail(error_code::number_after_minus); + } + const auto int_digits = static_cast(p - int_start); + std::size_t frac_digits = 0; + bool is_float = false; + if (p != e && *p == '.') + { + ++p; + const unsigned char* const f0 = p; + p = skip_digits(p, e); + if (NLOHMANN_VIEW_UNLIKELY(p == f0)) + { + return fail(error_code::number_after_dot); + } + frac_digits = static_cast(p - f0); + is_float = true; + } + std::int64_t exponent = 0; + if (p != e && (*p | 0x20) == 'e') + { + ++p; + bool exp_negative = false; + if (p != e && (*p == '+' || *p == '-')) + { + exp_negative = *p == '-'; + ++p; + } + if (NLOHMANN_VIEW_UNLIKELY(p == e || !is_digit(*p))) + { + return fail(error_code::number_after_exponent); + } + while (p != e && is_digit(*p)) + { + if (exponent < 100000) + { + exponent = (exponent * 10) + (*p - '0'); + } + ++p; + } + if (exp_negative) + { + exponent = -exponent; + } + is_float = true; + } + + value_t kind = value_t::number_float; + if (!is_float) + { + kind = negative ? value_t::number_integer : value_t::number_unsigned; + } + if (!is_float) + { + // integers that do not fit become floats, as in parse() + if (NLOHMANN_VIEW_UNLIKELY(int_digits >= 19)) + { + if (negative) + { + if (int_digits > 19 || (int_digits == 19 && digits_exceed(int_start, "9223372036854775808", 19))) + { + kind = value_t::number_float; + } + } + else if (int_digits > 20 || (int_digits == 20 && digits_exceed(int_start, "18446744073709551615", 20))) + { + kind = value_t::number_float; + } + } + } + // parse() rejects floats that overflow; only numbers whose magnitude + // could reach the largest FloatType (1e308 for double, 1e38 for + // float) need the conversion + if (NLOHMANN_VIEW_UNLIKELY(static_cast(int_digits) + exponent > std::numeric_limits::max_exponent10 - 8 && kind == value_t::number_float)) + { + if (builder::float_overflows(s, p)) + { + p = s; + return fail(error_code::number_overflow); + } + } + const auto layout = static_cast((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8)); + auto second = static_cast(p - s); + if (kind != value_t::number_float) + { + // integers are converted now, while their digits are in cache + const std::uint64_t m = int_digits <= 19 ? parse_upto19(int_start, static_cast(int_digits), e) + : (parse_upto19(int_start, 19, e) * 10) + static_cast(int_start[19] - '0'); + second = negative ? 0 - m : m; + } + emit(kind, 0, layout, static_cast(s - b), second); + return true; + } + + /// a string (value or key) at p + NLOHMANN_VIEW_ALWAYS_INLINE bool string() + { + ++p; // opening quote + const unsigned char* const s = p; + p = scan_string_run(p, e); + if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"')) + { + emit(value_t::string, 0, 0, static_cast(s - b), static_cast(p - s)); + ++p; + return true; + } + const decoded r = cold.slow_string(s, p); + if (r.p == nullptr) + { + return false; + } + p = r.p; + emit(value_t::string, node_flags::escaped, 0, r.start, r.len); + return true; + } + }; +}; + +/// run the builder with compile-time options +template +inline bool build_with(document_data& d, const char* src, std::size_t size, bool sentinel, parse_failure& failure) +{ + if (sentinel) + { + builder bld(d, src, size); + const bool ok = bld.run(); + failure = bld.failure(); + return ok; + } + builder bld(d, src, size); + const bool ok = bld.run(); + failure = bld.failure(); + return ok; +} + +/// sentinel: src[size] is readable and 0 (e.g. std::string); FloatType: the +/// number_float_t of the document +template +inline bool build(document_data& d, const char* src, std::size_t size, bool comments, bool trailing_commas, bool sentinel, parse_failure& failure) +{ + if (comments) + { + return trailing_commas ? build_with(d, src, size, sentinel, failure) + : build_with(d, src, size, sentinel, failure); + } + return trailing_commas ? build_with(d, src, size, sentinel, failure) + : build_with(d, src, size, sentinel, failure); +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/document_data.hpp b/include/nlohmann/detail/view/document_data.hpp new file mode 100644 index 000000000..d74383a85 --- /dev/null +++ b/include/nlohmann/detail/view/document_data.hpp @@ -0,0 +1,130 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // size_t +#include // memcpy +#include // operator new, placement new +#include // string + +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// storage of a parsed document; heap-allocated (header and an initial node +/// array in one block) so that views survive moves of the owning document +struct document_data +{ + const char* src = nullptr; + std::size_t size = 0; + node* tape = nullptr; + std::size_t tape_size = 0; + std::size_t tape_cap = 0; + node* inline_tape = nullptr; ///< node array allocated together with this header + std::size_t inline_cap = 0; + std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init) + std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init) + std::array base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage) + bool discarded = true; + + /// one allocation for the header and room for `nodes` nodes; large + /// documents get a separate node array instead (so it can be trimmed) + static document_data* create(std::size_t nodes) + { + nodes = nodes <= 256 ? nodes : 0; + void* mem = ::operator new (sizeof(document_data) + (nodes * sizeof(node))); + auto* d = new (mem) document_data(); // NOLINT(cppcoreguidelines-owning-memory): owned by the returned pointer, freed by deleter + // (aligned: sizeof is a multiple of the alignment; through void*, as GCC's -Wcast-align wants) + d->inline_tape = static_cast(static_cast(static_cast(mem) + sizeof(document_data))); // NOLINT(bugprone-casting-through-void) + d->inline_cap = nodes; + d->tape = d->inline_tape; + d->tape_cap = nodes; + return d; + } + + struct deleter + { + void operator()(document_data* d) const noexcept + { + d->~document_data(); + ::operator delete (d); + } + }; + + document_data() noexcept = default; + document_data(const document_data&) = delete; + document_data(document_data&&) = delete; + document_data& operator=(const document_data&) = delete; + document_data& operator=(document_data&&) = delete; + ~document_data() + { + release(); + } + + void release() noexcept + { + if (tape != inline_tape) + { + ::operator delete (tape); + } + tape = inline_tape; + tape_cap = inline_cap; + } + + /// make room for n nodes; keeps the first tape_size nodes + void reserve(std::size_t n) + { + if (n <= tape_cap) + { + return; + } + node* fresh = static_cast(::operator new (n * sizeof(node))); + if (tape_size != 0) + { + std::memcpy(fresh, tape, tape_size * sizeof(node)); + } + release(); + tape = fresh; + tape_cap = n; + } + + const char* str(const node& n) const noexcept + { + return base[n.flags & node_flags::storage] + n.off; + } + + /// the node after n's subtree (containers span `next` nodes, scalars one) + static NLOHMANN_VIEW_ALWAYS_INLINE const node* after(const node* n) noexcept + { + return n + (is_container(*n) ? n->next : 1u); + } + + /// first element (array) or first key (object) of a container + static NLOHMANN_VIEW_ALWAYS_INLINE const node* first_child(const node* n) noexcept + { + return n + 1; + } + + /// end of the elements of a container + static NLOHMANN_VIEW_ALWAYS_INLINE const node* child_end(const node* n) noexcept + { + return n + n->next; + } +}; + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/errors.hpp b/include/nlohmann/detail/view/errors.hpp new file mode 100644 index 000000000..ff5816efd --- /dev/null +++ b/include/nlohmann/detail/view/errors.hpp @@ -0,0 +1,85 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // min +#include // size_t +#include // string + +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +// Exceptions are thrown out of line, so that the accessors that may throw stay +// small enough to be inlined. + +[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_type_error(int id, const char* prefix, const char* type) +{ + NLOHMANN_VIEW_THROW(type_error::create(id, concat(prefix, type), nullptr)); +} + +[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_out_of_range(int id, const std::string& msg) +{ + NLOHMANN_VIEW_THROW(out_of_range::create(id, msg, nullptr)); +} + +[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_invalid_iterator(int id, const char* msg) +{ + NLOHMANN_VIEW_THROW(invalid_iterator::create(id, msg, nullptr)); +} + +/*! +@brief throw the exception BasicJsonType::parse would throw for this input + +The view accepts exactly the inputs parse() accepts, so on a failure the +library parser is run on the same bytes: it throws the exception parse() would +throw, with the same message, position, and "last read" token. The error path +is cold, so this costs nothing on valid input. Should parse() accept the input +nevertheless (a bug), the view's own failure is reported. +*/ +template +[[noreturn]] NLOHMANN_VIEW_NOINLINE void throw_parse_failure(const parse_failure& f, const char* src, std::size_t size, + bool ignore_comments, bool ignore_trailing_commas) +{ + if (f.code == error_code::input_too_large) + { + // LCOV_EXCL_START (4 GiB) + NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr)); + // LCOV_EXCL_STOP + } + const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas); + // LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug) + static_cast(accepted); + + position_t pos; + const std::size_t off = (std::min)(f.offset, size); + pos.chars_read_total = off + 1; + std::size_t line_start = 0; + for (std::size_t i = 0; i < off; ++i) + { + if (src[i] == '\n') + { + ++pos.lines_read; + line_start = i + 1; + } + } + pos.chars_read_current_line = off + 1 - line_start; + NLOHMANN_VIEW_THROW(parse_error::create(101, pos, "syntax error while parsing value", nullptr)); + // LCOV_EXCL_STOP +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/input.hpp b/include/nlohmann/detail/view/input.hpp new file mode 100644 index 000000000..e8a5e3148 --- /dev/null +++ b/include/nlohmann/detail/view/input.hpp @@ -0,0 +1,88 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // basic_string, char_traits, string +#include // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference +#include // forward + +#include +#include + +#if NLOHMANN_VIEW_HAS_CPP_17 + #include // string_view +#endif + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// how a document takes its input +enum class input_kind +{ + move_string, ///< rvalue std::string: owned without a copy + c_string, ///< const char* (NUL-terminated): borrowed + char_array, ///< char array (e.g. a string literal): borrowed + borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed + copy_range, ///< rvalue contiguous byte container: copied + adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer +}; + +template +struct classify_input +{ + using R = typename std::remove_reference::type; + using D = typename std::decay::type; + static constexpr bool is_rvalue = !std::is_lvalue_reference::value; + static constexpr bool is_bytes = is_contiguous_byte_container::value; +#if NLOHMANN_VIEW_HAS_CPP_17 + static constexpr bool is_string_view = std::is_same::value; +#else + static constexpr bool is_string_view = false; +#endif + // NOLINTBEGIN(readability-avoid-nested-conditional-operator): a constant expression of C++11 + static constexpr input_kind value = + std::is_array::value ? input_kind::char_array + : std::is_pointer::value ? input_kind::c_string + : (is_rvalue && std::is_same::value) ? input_kind::move_string + : (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range + : is_bytes ? input_kind::copy_range + : input_kind::adapter; + // NOLINTEND(readability-avoid-nested-conditional-operator) +}; + +/// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel) +template +struct is_std_string : std::false_type {}; + +template +struct is_std_string> : std::true_type {}; + +/// drain a json input adapter (UTF-16/32 inputs arrive as UTF-8) +template +std::string collect_adapter(Adapter ia) +{ + std::string buf; + for (;;) + { + const auto ch = ia.get_character(); + if (ch == std::char_traits::eof()) + { + break; + } + buf.push_back(static_cast(ch)); + } + return buf; +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/macro_scope.hpp b/include/nlohmann/detail/view/macro_scope.hpp new file mode 100644 index 000000000..8bbae1970 --- /dev/null +++ b/include/nlohmann/detail/view/macro_scope.hpp @@ -0,0 +1,71 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +// Macros of json_view.hpp and its detail headers. json.hpp undefines its own +// macros at its end (macro_unscope.hpp), so the view defines the few it needs +// under its own prefix; json_view.hpp undefines them all at its end +// (detail/view/macro_unscope.hpp). Configuration that json.hpp undefines is +// read from detail::abi_config instead. + +#if (defined(__cplusplus) && __cplusplus >= 201703L) || (defined(_MSVC_LANG) && _MSVC_LANG >= 201703L) + #define NLOHMANN_VIEW_HAS_CPP_17 1 +#else + #define NLOHMANN_VIEW_HAS_CPP_17 0 +#endif + +#if defined(__GNUC__) || defined(__clang__) + #define NLOHMANN_VIEW_LIKELY(x) __builtin_expect(!!(x), 1) + #define NLOHMANN_VIEW_UNLIKELY(x) __builtin_expect(!!(x), 0) + #define NLOHMANN_VIEW_ALWAYS_INLINE inline __attribute__((always_inline)) + #define NLOHMANN_VIEW_NOINLINE __attribute__((noinline)) +#elif defined(_MSC_VER) + #define NLOHMANN_VIEW_LIKELY(x) (x) + #define NLOHMANN_VIEW_UNLIKELY(x) (x) + #define NLOHMANN_VIEW_ALWAYS_INLINE __forceinline + #define NLOHMANN_VIEW_NOINLINE __declspec(noinline) +#else + #define NLOHMANN_VIEW_LIKELY(x) (x) + #define NLOHMANN_VIEW_UNLIKELY(x) (x) + #define NLOHMANN_VIEW_ALWAYS_INLINE inline + #define NLOHMANN_VIEW_NOINLINE +#endif + +#if defined(__GNUC__) || defined(__clang__) + #define NLOHMANN_VIEW_NODISCARD __attribute__((warn_unused_result)) +#elif defined(_MSC_VER) + #define NLOHMANN_VIEW_NODISCARD _Check_return_ +#else + #define NLOHMANN_VIEW_NODISCARD +#endif + +// exceptions as in json.hpp (JSON_NOEXCEPTION, JSON_THROW_USER) +#if (defined(__cpp_exceptions) || defined(__EXCEPTIONS) || defined(_CPPUNWIND)) && !defined(JSON_NOEXCEPTION) + #define NLOHMANN_VIEW_THROW(exception) throw exception +#else + #include + // (the exception is built first, so that the arguments of the throwing + // helpers count as used; the program ends anyway) + #define NLOHMANN_VIEW_THROW(exception) (static_cast(exception), std::abort()) +#endif +#if defined(JSON_THROW_USER) + #undef NLOHMANN_VIEW_THROW + #define NLOHMANN_VIEW_THROW JSON_THROW_USER +#endif + +// the parser stores a node's first word at once where the layout of `node` is +// known to be little-endian (MSVC targets are); elsewhere field by field +#if (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__) || defined(_MSC_VER) + #define NLOHMANN_VIEW_LITTLE_ENDIAN 1 +#else + #define NLOHMANN_VIEW_LITTLE_ENDIAN 0 +#endif + +/// sixteen checks at fixed offsets 0..15 +#define NLOHMANN_VIEW_REPEAT16(X) X(0) X(1) X(2) X(3) X(4) X(5) X(6) X(7) X(8) X(9) X(10) X(11) X(12) X(13) X(14) X(15) diff --git a/include/nlohmann/detail/view/macro_unscope.hpp b/include/nlohmann/detail/view/macro_unscope.hpp new file mode 100644 index 000000000..11f71ad1b --- /dev/null +++ b/include/nlohmann/detail/view/macro_unscope.hpp @@ -0,0 +1,21 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +// undefine the macros of detail/view/macro_scope.hpp (at the end of json_view.hpp) + +#undef NLOHMANN_VIEW_HAS_CPP_17 +#undef NLOHMANN_VIEW_LIKELY +#undef NLOHMANN_VIEW_UNLIKELY +#undef NLOHMANN_VIEW_ALWAYS_INLINE +#undef NLOHMANN_VIEW_NOINLINE +#undef NLOHMANN_VIEW_NODISCARD +#undef NLOHMANN_VIEW_THROW +#undef NLOHMANN_VIEW_LITTLE_ENDIAN +#undef NLOHMANN_VIEW_REPEAT16 diff --git a/include/nlohmann/detail/view/materialize.hpp b/include/nlohmann/detail/view/materialize.hpp new file mode 100644 index 000000000..e9ffe0668 --- /dev/null +++ b/include/nlohmann/detail/view/materialize.hpp @@ -0,0 +1,131 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // int64_t, uint8_t +#include // string +#include // vector + +#include +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/*! +@brief the basic_json value of the subtree at n + +The subtree is replayed into the SAX handler that parse() uses to build its +values, so the result is the value parse() would produce: duplicate keys keep +the last value, and with JSON_DIAGNOSTICS the parent pointers are set. It is +iterative, so the nesting depth is limited by memory only, as for parse(). +Without a lexer the handler records no source positions +(JSON_DIAGNOSTIC_POSITIONS). +*/ +template +BasicJsonType materialize(const document_data& d, const node* n) +{ + using string_t = typename BasicJsonType::string_t; + using sax_t = json_sax_dom_parser>; + + BasicJsonType result; + sax_t sax(result, true); + const string_t no_token{}; + // the ends of the open containers, and whether they are objects + std::vector> open; + for (;;) + { + switch (static_cast(n->kind)) + { + case value_t::object: + case value_t::array: + { + const bool object = n->kind == static_cast(value_t::object); + if (object) + { + sax.start_object(n->len); + } + else + { + sax.start_array(n->len); + } + open.emplace_back(document_data::child_end(n), object); + n = document_data::first_child(n); + break; + } + case value_t::string: + { + string_t s(d.str(*n), n->len); + sax.string(s); + ++n; + break; + } + case value_t::number_integer: + sax.number_integer(static_cast(static_cast(integer_bits(*n)))); + ++n; + break; + case value_t::number_unsigned: + sax.number_unsigned(static_cast(integer_bits(*n))); + ++n; + break; + case value_t::number_float: + sax.number_float(float_value(d.str(*n), *n), no_token); + ++n; + break; + case value_t::boolean: + sax.boolean((n->flags & node_flags::is_true) != 0); + ++n; + break; + case value_t::null: + case value_t::binary: + case value_t::discarded: + default: + sax.null(); + ++n; + break; + } + for (;;) + { + if (open.empty()) + { + return result; + } + if (n != open.back().first) + { + break; + } + if (open.back().second) + { + sax.end_object(); + } + else + { + sax.end_array(); + } + open.pop_back(); + } + if (open.back().second) + { + // the key of the next member + string_t key(d.str(*n), n->len); + sax.key(key); + ++n; + } + } +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/node.hpp b/include/nlohmann/detail/view/node.hpp new file mode 100644 index 000000000..471c1f844 --- /dev/null +++ b/include/nlohmann/detail/view/node.hpp @@ -0,0 +1,97 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t +#include // uint8_t, uint16_t, uint32_t, uint64_t +#include // memcpy + +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +// the node kinds are value_t values; the tests of is_container() and of the +// number kinds depend on this numbering +static_assert(static_cast(value_t::null) == 0 && static_cast(value_t::object) == 1 + && static_cast(value_t::array) == 2 && static_cast(value_t::string) == 3 + && static_cast(value_t::boolean) == 4 && static_cast(value_t::number_integer) == 5 + && static_cast(value_t::number_unsigned) == 6 && static_cast(value_t::number_float) == 7, + "the node format depends on the numbering of value_t"); + +/// node flags +struct node_flags +{ + static constexpr std::uint8_t escaped = 1; ///< string payload lives in the decode arena, not the source + static constexpr std::uint8_t storage = 3; ///< mask: where a string or number token lives (index into document_data::base) + static constexpr std::uint8_t is_true = 4; ///< boolean value +}; + +/// One entry of the flat index, in document order. An object's members are +/// stored as key node followed by the value's subtree. Integers keep their +/// converted 64-bit value in the len/next bytes (the node after a scalar is +/// always the next one, and the token length follows from `extra`). +struct node +{ + std::uint8_t kind; ///< value_t + std::uint8_t flags; ///< node_flags + std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; otherwise 0 + std::uint32_t off; ///< source offset (string content, number token, literal, bracket); arena offset if node_flags::escaped + std::uint32_t len; ///< string: decoded bytes; float: token bytes; array/object: element count + std::uint32_t next; ///< array/object: number of nodes of the subtree (its extent in the enclosing sequence) +}; +static_assert(sizeof(node) == 16, "node must stay 16 bytes"); + +NLOHMANN_VIEW_ALWAYS_INLINE bool is_container(const node& n) noexcept +{ + return static_cast(n.kind) - 1u <= 1u; +} + +/// the converted value of an integer node (stored in len/next) +NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept +{ + std::uint64_t v = 0; + std::memcpy(&v, reinterpret_cast(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return v; +} + +NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept +{ + std::memcpy(reinterpret_cast(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +} + +/// token length of a number node +NLOHMANN_VIEW_ALWAYS_INLINE std::uint32_t number_length(const node& n) noexcept +{ + return n.kind == static_cast(value_t::number_float) ? n.len + : (n.extra & 0xFFu) + (n.kind == static_cast(value_t::number_integer) ? 1u : 0u); +} + +/// estimated number of nodes for an input of `size` bytes (one node per ~12 +/// bytes covers typical documents without regrowth) +inline std::size_t estimate_nodes(std::size_t size) noexcept +{ + return (size / 12) + 16; +} + +/// estimated number of nodes for the input [src, src + size): pretty-printed +/// input (whitespace after the first byte) needs about a node per 12 bytes, +/// minified input up to one per 4 (yyjson tells the two apart the same way) +inline std::size_t estimate_nodes(const char* src, std::size_t size) noexcept +{ + return size >= 2 && (src[1] == ' ' || src[1] == '\n' || src[1] == '\r' || src[1] == '\t') ? estimate_nodes(size) : (size / 4) + 16; +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/number.hpp b/include/nlohmann/detail/view/number.hpp new file mode 100644 index 000000000..c9befdd49 --- /dev/null +++ b/include/nlohmann/detail/view/number.hpp @@ -0,0 +1,67 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // size_t +#include // string + +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/*! +@brief the value of the float token of a node, as parse() converts it + +Uses the lexer's conversion (detail::convert_float), so that the values are +bit-identical to parse(): float and double are converted without allocation +and independent of the locale. The digit layout recorded while parsing locates +the decimal point and the exponent without scanning the token. +*/ +template +NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) +{ + const char* const last = first + n.len; + const std::size_t neg = first[0] == '-' ? 1 : 0; + const std::size_t int_digits = n.extra & 0xFFu; + const std::size_t frac_digits = n.extra >> 8u; + std::size_t dot = std::string::npos; + std::size_t mantissa_end = n.len; + if (int_digits != 255 && frac_digits != 255) + { + dot = frac_digits != 0 ? neg + int_digits : std::string::npos; + mantissa_end = neg + int_digits + (frac_digits != 0 ? 1 + frac_digits : 0); + } + else + { + // more digits than the layout records: locate them + for (std::size_t i = 0; i < n.len; ++i) + { + if (first[i] == '.') + { + dot = i; + } + else if (first[i] == 'e' || first[i] == 'E') + { + mantissa_end = i; + break; + } + } + } + return convert_float(first, last, dot, mantissa_end); +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/scan.hpp b/include/nlohmann/detail/view/scan.hpp new file mode 100644 index 000000000..6d732ae36 --- /dev/null +++ b/include/nlohmann/detail/view/scan.hpp @@ -0,0 +1,189 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2020 YaoYuan +// SPDX-License-Identifier: MIT + +#pragma once + +#include // array +#include // size_t +#include // uint8_t, uint16_t, uint64_t +#include // memcpy + +#include +#include + +// Scanning primitives of the view's parser. The unrolled checks at fixed +// offsets follow yyjson (https://github.com/ibireme/yyjson, MIT license): the +// loads do not depend on each other, so the CPU can run ahead. Words are read +// with read_eight_bytes(), so nothing here depends on the byte order. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// 1 for bytes that may appear verbatim in a string: 0x20..0x7F except '"' and '\\' +inline const std::uint8_t* string_plain() noexcept +{ + static const std::array table = + { + { + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x00..0x1F + 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, // 0x20..0x3F ('"') + 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, // 0x40..0x5F ('\\') + 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, // 0x60..0x7F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x80..0x9F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xA0..0xBF + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xC0..0xDF + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xE0..0xFF + } + }; + return table.data(); +} + +NLOHMANN_VIEW_ALWAYS_INLINE bool is_digit(unsigned char c) noexcept +{ + return static_cast(c - '0') <= 9; +} + +/// two bytes as they are in memory (only compared with byte-symmetric patterns) +NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcept +{ + std::uint16_t w = 0; + std::memcpy(&w, p, 2); + return w; +} + +/// Advance over plain string bytes and well-formed UTF-8. Stops at a quote, +/// a backslash, a control character, ill-formed UTF-8, or the end. The first +/// 16 bytes are checked one by one, so that the position advances by +/// constants in predicted branches (most strings are short); longer runs +/// continue eight bytes at a time. +NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept +{ + const std::uint8_t* plain = string_plain(); + for (;;) + { + if (e - p >= 16) + { +#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; } + NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) +#undef NLOHMANN_VIEW_STEP + p += 16; + while (e - p >= 8) + { + const std::uint64_t special = swar_string_special(read_eight_bytes(p)); + if (special != 0) + { + p += count_trailing_zeros(special) / 8; + goto stop; + } + p += 8; + } + continue; + } + while (p != e && plain[*p] != 0) + { + ++p; + } + if (p == e) + { + return p; + } +stop: + if (*p < 0x80) + { + return p; // quote, backslash, or control character + } + // non-ASCII: a run of well-formed sequences (the library's check, so + // that exactly what json::parse accepts is accepted) + do + { + const std::size_t n = validate_one_utf8(p, static_cast(e - p)); + if (n == 0) + { + return p; + } + p += n; + } + while (p != e && *p >= 0x80); + } +} + +/// advance over ASCII digits +NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* skip_digits(const unsigned char* p, const unsigned char* e) noexcept +{ + while (e - p >= 16) + { +#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(is_digit(p[i]))) {} else { return p + (i); } + NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) +#undef NLOHMANN_VIEW_STEP + p += 16; + } + while (p != e && is_digit(*p)) + { + ++p; + } + return p; +} + +/// powers of ten up to 10^19 as integers +inline std::uint64_t int_pow10(unsigned k) noexcept +{ + static const std::array table = + { + { + 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, + 10000000000u, 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, + 10000000000000000u, 100000000000000000u, 1000000000000000000u, 10000000000000000000u + } + }; + return table[k]; +} + +/// value of 0 < k < 8 digits at p in one step if [p, p + 8) lies below +/// limit, else one digit at a time (whole blocks of eight digits are read by +/// parse_upto19() directly) +NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t parse_upto8(const unsigned char* p, unsigned k, const unsigned char* limit) noexcept +{ + if (NLOHMANN_VIEW_LIKELY(limit - p >= 8)) + { + // move the k digits to the top and pad the vacated low bytes with '0' + const unsigned shift = 8 * (8 - k); + return parse_eight_digits((read_eight_bytes(p) << shift) | (0x3030303030303030u >> (8 * k))); + } + std::uint64_t v = 0; + for (unsigned i = 0; i < k; ++i) + { + v = (v * 10) + static_cast(p[i] - '0'); + } + return v; +} + +/// value of k <= 19 digits at p +NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t parse_upto19(const unsigned char* p, unsigned k, const unsigned char* limit) noexcept +{ + std::uint64_t w = 0; + while (k >= 8) + { + // (eight digits of the token: they lie below limit) + w = (w * 100000000u) + parse_eight_digits(read_eight_bytes(p)); + p += 8; + k -= 8; + } + if (k != 0) + { + w = (w * int_pow10(k)) + parse_upto8(p, k, limit); + } + return w; +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/detail/view/string_ref.hpp b/include/nlohmann/detail/view/string_ref.hpp new file mode 100644 index 000000000..d401500ec --- /dev/null +++ b/include/nlohmann/detail/view/string_ref.hpp @@ -0,0 +1,113 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#pragma once + +#include // min +#include // size_t +#include // memcmp, strlen +#include // basic_string + +#include +#include + +#if NLOHMANN_VIEW_HAS_CPP_17 + #include // string_view +#endif +#ifndef JSON_NO_IO + #include // ostream +#endif + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +#if NLOHMANN_VIEW_HAS_CPP_17 +using string_ref = std::string_view; +#else +/// minimal C++11 stand-in for std::string_view +class string_ref +{ + public: + using size_type = std::size_t; + using const_iterator = const char*; + + string_ref() noexcept = default; + // s must be null-terminated, as for std::string_view(const char*) + // flawfinder: ignore + string_ref(const char* s) : m_data(s), m_size(std::strlen(s)) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + string_ref(const char* s, std::size_t n) noexcept : m_data(s), m_size(n) {} + template + string_ref(const std::basic_string& s) noexcept : m_data(s.data()), m_size(s.size()) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + + const char* data() const noexcept + { + return m_data; + } + std::size_t size() const noexcept + { + return m_size; + } + std::size_t length() const noexcept + { + return m_size; + } + bool empty() const noexcept + { + return m_size == 0; + } + const char* begin() const noexcept + { + return m_data; + } + const char* end() const noexcept + { + return m_data + m_size; + } + char operator[](std::size_t i) const noexcept + { + return m_data[i]; + } + + template + explicit operator std::basic_string() const + { + return std::basic_string(m_data, m_size); + } + + friend bool operator==(string_ref a, string_ref b) noexcept + { + return a.m_size == b.m_size && (a.m_size == 0 || std::memcmp(a.m_data, b.m_data, a.m_size) == 0); + } + friend bool operator!=(string_ref a, string_ref b) noexcept + { + return !(a == b); + } + friend bool operator<(string_ref a, string_ref b) noexcept + { + const int c = std::memcmp(a.m_data, b.m_data, (std::min)(a.m_size, b.m_size)); + return c != 0 ? c < 0 : a.m_size < b.m_size; + } +#ifndef JSON_NO_IO + friend std::ostream& operator<<(std::ostream& o, string_ref s) + { + return o.write(s.m_data, static_cast(s.m_size)); + } +#endif + + private: + const char* m_data = ""; + std::size_t m_size = 0; +}; +#endif + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END diff --git a/include/nlohmann/json.hpp b/include/nlohmann/json.hpp index aa8b1a6c6..2f5da27ba 100644 --- a/include/nlohmann/json.hpp +++ b/include/nlohmann/json.hpp @@ -44,6 +44,7 @@ // translation unit that relies on basic_json<>'s defaults actually being usable. #include // IWYU pragma: keep #include +#include #include #include // IWYU pragma: keep #include // IWYU pragma: keep diff --git a/include/nlohmann/json_view.hpp b/include/nlohmann/json_view.hpp new file mode 100644 index 000000000..5aec36ef9 --- /dev/null +++ b/include/nlohmann/json_view.hpp @@ -0,0 +1,595 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +/****************************************************************************\ + * Zero-copy, read-only view of a parsed JSON text. * + * * + * json_document::parse() builds a flat index of the values of a JSON text * + * (16 bytes per value) instead of a tree of basic_json values. Strings and * + * numbers stay in the source text; only strings with escapes are decoded, * + * into one buffer. json_view is a handle to one value of the document, with * + * the read-only part of the basic_json interface; materialize() turns a * + * subtree into the basic_json value that parse() would produce. * + * * + * The source text must outlive a document that borrows it (lvalue byte * + * containers, C strings); rvalue strings, streams, and other inputs are * + * owned by the document. * +\****************************************************************************/ + +#ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_ +#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ + +#include // size_t +#include // memcpy, strlen +#include // distance, input_iterator_tag, iterator_traits +#include // unique_ptr +#include // string +#include // enable_if, integral_constant, is_base_of, is_integral, is_same, remove_cv, remove_extent +#include // forward, move + +#include + +// the view builds on internals of the library: both must be the same version +#if NLOHMANN_JSON_VERSION_MAJOR != 3 || NLOHMANN_JSON_VERSION_MINOR != 12 || NLOHMANN_JSON_VERSION_PATCH != 0 + #error "json_view.hpp requires json.hpp of the same version (3.12.0)" +#endif + +#include +#include +#include +#include +#include +#include +#include +#include + +NLOHMANN_JSON_NAMESPACE_BEGIN + +template +class basic_json_document; + +/*! +@brief read-only handle to one value of a basic_json_document + +Trivially copyable (two pointers). Valid as long as the document is alive and +has not been re-parsed, and as long as a borrowed source text is alive. +*/ +template +class basic_json_view +{ + using node = detail::view::node; + using document_data = detail::view::document_data; + + public: + using value_t = detail::value_t; + using string_t = typename BasicJsonType::string_t; + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using json_pointer = typename BasicJsonType::json_pointer; + using size_type = std::size_t; + /// std::string_view from C++17 on + using string_view_t = detail::view::string_ref; + + /// an invalid view (type() == value_t::discarded) + basic_json_view() noexcept = default; + + ////////// + // type // + ////////// + + NLOHMANN_VIEW_ALWAYS_INLINE value_t type() const noexcept + { + return m_node != nullptr ? static_cast(m_node->kind) : value_t::discarded; + } + + bool is_null() const noexcept + { + return type() == value_t::null; + } + + bool is_boolean() const noexcept + { + return type() == value_t::boolean; + } + + bool is_number() const noexcept + { + return is_number_integer() || is_number_float(); + } + + bool is_number_integer() const noexcept + { + return type() == value_t::number_integer || type() == value_t::number_unsigned; + } + + bool is_number_unsigned() const noexcept + { + return type() == value_t::number_unsigned; + } + + bool is_number_float() const noexcept + { + return type() == value_t::number_float; + } + + bool is_string() const noexcept + { + return type() == value_t::string; + } + + bool is_array() const noexcept + { + return type() == value_t::array; + } + + bool is_object() const noexcept + { + return type() == value_t::object; + } + + /// always false: JSON text has no binary values + bool is_binary() const noexcept + { + return false; + } + + bool is_primitive() const noexcept + { + return is_null() || is_string() || is_boolean() || is_number(); + } + + bool is_structured() const noexcept + { + return is_array() || is_object(); + } + + /// the root of a failed parse with allow_exceptions == false, or a + /// default-constructed view + bool is_discarded() const noexcept + { + return type() == value_t::discarded; + } + + /// false for discarded views + explicit operator bool() const noexcept + { + return m_node != nullptr; + } + + ////////////// + // capacity // + ////////////// + + /// the number of elements (arrays, objects), 0 for null and discarded, + /// 1 otherwise, as basic_json::size() + size_type size() const noexcept + { + switch (type()) + { + case value_t::null: + case value_t::discarded: + return 0; + case value_t::array: + case value_t::object: + return m_node->len; + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + default: + return 1; + } + } + + /// as basic_json::empty() + bool empty() const noexcept + { + switch (type()) + { + case value_t::null: + case value_t::discarded: + return true; + case value_t::array: + case value_t::object: + return m_node->len == 0; + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + default: + return false; + } + } + + ///////////////// + // materialize // + ///////////////// + + /// the basic_json value of this subtree, as parse() would produce it + /// (a discarded value for a discarded view) + BasicJsonType materialize() const + { + if (m_node == nullptr) + { + return BasicJsonType(value_t::discarded); + } + return detail::view::materialize(*m_doc, m_node); + } + + /// byte offset of this value in the source text (for strings: of the + /// first byte after the opening quote); static_cast(-1) for + /// a discarded view and for strings with escapes, which are decoded + std::size_t source_offset() const noexcept + { + return m_node != nullptr && (m_node->flags & detail::view::node_flags::storage) == 0 + ? m_node->off : static_cast(-1); + } + + private: + template friend class basic_json_document; + + basic_json_view(const document_data* d, const node* n) noexcept + : m_doc(d), m_node(n) + {} + + const document_data* m_doc = nullptr; + const node* m_node = nullptr; +}; + +/*! +@brief a parsed JSON text: owns the node index (and, optionally, the text) + +Borrowed parses keep a pointer to the caller's text, which must outlive the +document. Owned parses (parse_copy, rvalue std::string, streams, and inputs +that are not contiguous byte ranges) keep their own copy. +*/ +template +class basic_json_document +{ + using document_data = detail::view::document_data; + + static_assert(sizeof(typename BasicJsonType::number_integer_t) == 8 && sizeof(typename BasicJsonType::number_unsigned_t) == 8, + "json_view supports 64-bit integer types only"); + + public: + using view_type = basic_json_view; + using value_t = detail::value_t; + + /// an empty (discarded) document + basic_json_document() = default; + basic_json_document(basic_json_document&&) noexcept = default; + basic_json_document& operator=(basic_json_document&&) noexcept = default; + basic_json_document(const basic_json_document&) = delete; + basic_json_document& operator=(const basic_json_document&) = delete; + ~basic_json_document() = default; + + ///////////// + // parsing // + ///////////// + + /// parse a JSON text; contiguous byte inputs are borrowed, everything else + /// (and rvalue std::string) is owned + template + NLOHMANN_VIEW_NODISCARD + static basic_json_document parse(InputType&& input, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false) + { + basic_json_document d; + d.read(std::forward(input), allow_exceptions, ignore_comments, ignore_trailing_commas); + return d; + } + + /// parse [first, last) + template::iterator_category>::value, int>::type = 0> + NLOHMANN_VIEW_NODISCARD + static basic_json_document parse(IteratorType first, IteratorType last, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false) + { + basic_json_document d; + d.read_range(first, last, allow_exceptions, ignore_comments, ignore_trailing_commas); + return d; + } + + /// parse a copy of the input; the document does not depend on it afterwards + template + NLOHMANN_VIEW_NODISCARD + static basic_json_document parse_copy(InputType&& input, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false) + { + basic_json_document d; + d.build_owned(collect(std::forward(input)), allow_exceptions, ignore_comments, ignore_trailing_commas); + return d; + } + + /// check whether the input is valid JSON (the result of basic_json::accept) + template + static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false) + { + basic_json_document d; + d.read(std::forward(input), false, ignore_comments, ignore_trailing_commas); + return !d.is_discarded(); + } + + /// parse into this document, reusing its memory + template + // flawfinder: ignore (a member function, not POSIX read()) + void read(InputType&& input, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false) + { + read_kind(std::forward(input), allow_exceptions, ignore_comments, ignore_trailing_commas, + std::integral_constant::value> {}); + } + + //////////// + // access // + //////////// + + /// the root value (discarded if parsing failed without exceptions) + view_type root() const noexcept + { + if (!m_data || m_data->discarded) + { + return view_type(); + } + return view_type(m_data.get(), m_data->tape); + } + + bool is_discarded() const noexcept + { + return !m_data || m_data->discarded; + } + + /// the parsed text + typename view_type::string_view_t source() const noexcept + { + return m_data ? typename view_type::string_view_t(m_data->src, m_data->size) : typename view_type::string_view_t(); + } + + /// whether the document holds its own copy of the text + bool owns_source() const noexcept + { + return m_data && !m_data->owned.empty() && m_data->src == m_data->owned.data(); + } + + /// number of index nodes (values plus object keys) + std::size_t node_count() const noexcept + { + return m_data ? m_data->tape_size : 0; + } + + /// bytes held by the document (index, decoded strings, owned text) + std::size_t memory_usage() const noexcept + { + if (!m_data) + { + return 0; + } + return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node)) + + (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0) + + m_data->arena.capacity() + m_data->owned.capacity(); + } + + /// release unused capacity of the index and the decoded strings; like + /// std::vector::shrink_to_fit, this invalidates the views of the document + /// (take new ones from root()) + void shrink_to_fit() + { + if (!m_data) + { + return; + } + using detail::view::node; + document_data& d = *m_data; + + // allocate everything first, so that an exception leaves the document + // unchanged + const bool shrink_arena = d.arena.capacity() > d.arena.size(); + std::string arena(shrink_arena ? d.arena : std::string()); + const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap; + const bool into_header = d.tape_size <= d.inline_cap; + node* fresh = (shrink_tape && !into_header) ? static_cast(::operator new (d.tape_size * sizeof(node))) : d.inline_tape; + + if (shrink_tape) + { + std::memcpy(fresh, d.tape, d.tape_size * sizeof(node)); + ::operator delete (d.tape); + d.tape = fresh; + d.tape_cap = into_header ? d.inline_cap : d.tape_size; + } + if (shrink_arena) + { + d.arena.swap(arena); + d.base[1] = d.arena.data(); + } + } + + private: + using input_kind = detail::view::input_kind; + + /// create the storage (sized for the input) on first use + void ensure_data(const char* src, std::size_t size) + { + if (!m_data) + { + m_data.reset(document_data::create(detail::view::estimate_nodes(src, size))); + } + } + + /// parse a buffer the document takes ownership of + void build_owned(std::string&& buf, bool allow_exceptions, bool comments, bool trailing_commas) + { + ensure_data(buf.data(), buf.size()); + m_data->owned = std::move(buf); + build(m_data->owned.data(), m_data->owned.size(), allow_exceptions, comments, trailing_commas, true, true); + } + + /// sentinel: src[size] is readable and 0 (std::string, C strings) + void build(const char* src, std::size_t size, bool allow_exceptions, bool comments, bool trailing_commas, bool owned, bool sentinel) + { + ensure_data(src, size); + document_data& d = *m_data; + if (!owned) + { + d.owned.clear(); + } + d.src = src; + d.size = size; + d.tape_size = 0; + d.arena.clear(); + d.discarded = true; + detail::view::parse_failure failure; + bool ok = false; + if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u)) + { + failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB) + } + else + { + ok = detail::view::build < typename BasicJsonType::number_float_t, !detail::abi_config::strict_nul_handling > (d, src, size, comments, trailing_commas, sentinel, failure); + } + if (NLOHMANN_VIEW_LIKELY(ok)) + { + d.base[0] = d.src; + d.base[1] = d.arena.data(); + d.discarded = false; + return; + } + if (allow_exceptions) + { + detail::view::throw_parse_failure(failure, src, size, comments, trailing_commas); + } + } + + // --- input dispatch (see detail::view::input_kind) --- + + void read_kind(std::string&& s, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + build_owned(std::move(s), ae, c, tc); + } + + template + void read_kind(CharT* s, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + static_assert(sizeof(CharT) == 1 && std::is_integral::type>::value, "json_view parses byte (char-like) input"); + if (s == nullptr) + { + build("", 0, ae, c, tc, false, true); + return; + } + const char* cs = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + // C strings are null-terminated, as for json::parse(const char*) + // flawfinder: ignore + build(cs, std::strlen(cs), ae, c, tc, false, true); + } + + template + void read_kind(Array& a, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + using CharT = typename std::remove_cv::type>::type; + static_assert(sizeof(CharT) == 1 && std::is_integral::value, "json_view parses byte (char-like) input"); + const std::size_t n = std::extent::value; + // a trailing NUL (string literals) is not part of the text, as for + // parse(), and serves as sentinel + const bool terminated = n > 0 && a[n - 1] == 0 && (!detail::abi_config::strict_nul_handling || std::is_same::value); + build(reinterpret_cast(&a[0]), terminated ? n - 1 : n, ae, c, tc, false, terminated); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + template + void read_kind(const T& s, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + // std::basic_string guarantees data()[size()] == 0: use it as sentinel + const auto size = static_cast(s.size()); + build(size == 0 ? "" : reinterpret_cast(s.data()), size, ae, c, tc, false, // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + detail::view::is_std_string::value || size == 0); + } + + template + void read_kind(const T& s, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + build_owned(std::string(reinterpret_cast(s.data()), static_cast(s.size())), ae, c, tc); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + template + void read_kind(T&& input, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + build_owned(detail::view::collect_adapter(detail::input_adapter(std::forward(input))), ae, c, tc); + } + + template + void read_range(IteratorType first, IteratorType last, bool ae, bool c, bool tc) + { + using value_type = typename std::remove_cv::value_type>::type; + // borrow the range where the library's input adapter scans it as one + // block: pointers and, in C++20, contiguous iterators (std::vector, + // std::string, ...) over single bytes + read_range_impl(first, last, ae, c, tc, std::integral_constant < bool, std::is_integral::value + && detail::iterator_input_adapter::supports_bulk_scan > {}); + } + + template + void read_range_impl(IteratorType first, IteratorType last, bool ae, bool c, bool tc, std::true_type /*contiguous bytes*/) + { + const auto size = static_cast(std::distance(first, last)); + build(size == 0 ? "" : reinterpret_cast(&*first), size, ae, c, tc, false, size == 0); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + template + void read_range_impl(IteratorType first, IteratorType last, bool ae, bool c, bool tc, std::false_type /*other*/) + { + build_owned(detail::view::collect_adapter(detail::input_adapter(first, last)), ae, c, tc); + } + + template + static std::string collect(T&& input) + { + return collect_impl(std::forward(input), std::integral_constant::type>::value> {}); + } + + template + static std::string collect_impl(const T& s, std::true_type /*contiguous*/) + { + return {reinterpret_cast(s.data()), static_cast(s.size())}; // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + template + static std::string collect_impl(T&& s, std::false_type /*other*/) + { + return detail::view::collect_adapter(detail::input_adapter(std::forward(s))); + } + + std::unique_ptr m_data{}; // NOLINT(readability-redundant-member-init) +}; + +/// a parsed JSON text for json +using json_document = basic_json_document; +/// a value of a json_document +using json_view = basic_json_view; +/// a parsed JSON text for ordered_json +using ordered_json_document = basic_json_document; +/// a value of an ordered_json_document +using ordered_json_view = basic_json_view; + +NLOHMANN_JSON_NAMESPACE_END + +#include + +#endif // INCLUDE_NLOHMANN_JSON_VIEW_HPP_ diff --git a/meson.build b/meson.build index 2800957ec..2ccb0ef6e 100644 --- a/meson.build +++ b/meson.build @@ -16,6 +16,7 @@ if not meson.is_subproject() install_headers('single_include/nlohmann/json.hpp', subdir: 'nlohmann') install_headers('single_include/nlohmann/json_fwd.hpp', subdir: 'nlohmann') install_headers('single_include/nlohmann/json_literals.hpp', subdir: 'nlohmann') +install_headers('single_include/nlohmann/json_view.hpp', subdir: 'nlohmann') pkgc = import('pkgconfig') pkgc.generate(name: 'nlohmann_json', diff --git a/single_include/nlohmann/json.hpp b/single_include/nlohmann/json.hpp index c07bce33a..9a457d532 100644 --- a/single_include/nlohmann/json.hpp +++ b/single_include/nlohmann/json.hpp @@ -7545,6 +7545,43 @@ class byte_container_with_subtype : public BinaryType NLOHMANN_JSON_NAMESPACE_END +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ + +/*! +@brief the configuration macros that change the library's behavior + +json.hpp undefines these macros at its end (see macro_unscope.hpp), so code +that builds on the library after it (json_view.hpp) reads them here. Like the +macros, they are part of the ABI namespace, so they always match the +basic_json they are used with. +*/ +struct abi_config +{ + /// JSON_STRICT_NUL_HANDLING: a null byte is an error, not the end of input + static constexpr bool strict_nul_handling = JSON_STRICT_NUL_HANDLING != 0; + /// JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON + static constexpr bool legacy_discarded_value_comparison = JSON_USE_LEGACY_DISCARDED_VALUE_COMPARISON != 0; +}; + +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + // #include // #include diff --git a/single_include/nlohmann/json_view.hpp b/single_include/nlohmann/json_view.hpp new file mode 100644 index 000000000..b853fcd82 --- /dev/null +++ b/single_include/nlohmann/json_view.hpp @@ -0,0 +1,2649 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +/****************************************************************************\ + * Zero-copy, read-only view of a parsed JSON text. * + * * + * json_document::parse() builds a flat index of the values of a JSON text * + * (16 bytes per value) instead of a tree of basic_json values. Strings and * + * numbers stay in the source text; only strings with escapes are decoded, * + * into one buffer. json_view is a handle to one value of the document, with * + * the read-only part of the basic_json interface; materialize() turns a * + * subtree into the basic_json value that parse() would produce. * + * * + * The source text must outlive a document that borrows it (lvalue byte * + * containers, C strings); rvalue strings, streams, and other inputs are * + * owned by the document. * +\****************************************************************************/ + +#ifndef INCLUDE_NLOHMANN_JSON_VIEW_HPP_ +#define INCLUDE_NLOHMANN_JSON_VIEW_HPP_ + +#include // size_t +#include // memcpy, strlen +#include // distance, input_iterator_tag, iterator_traits +#include // unique_ptr +#include // string +#include // enable_if, integral_constant, is_base_of, is_integral, is_same, remove_cv, remove_extent +#include // forward, move + +#include + +// the view builds on internals of the library: both must be the same version +#if NLOHMANN_JSON_VERSION_MAJOR != 3 || NLOHMANN_JSON_VERSION_MINOR != 12 || NLOHMANN_JSON_VERSION_PATCH != 0 + #error "json_view.hpp requires json.hpp of the same version (3.12.0)" +#endif + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2020 YaoYuan +// SPDX-License-Identifier: MIT + + + +#include // find, find_if, max +#include // array +#include // size_t, ptrdiff_t +#include // int64_t, uint8_t, uint16_t, uint32_t, uint64_t +#include // memcmp, memcpy +#include // numeric_limits +#include // string +#include // vector + +// #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // array +#include // size_t +#include // memcpy +#include // operator new, placement new +#include // string + +// #include +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +// Macros of json_view.hpp and its detail headers. json.hpp undefines its own +// macros at its end (macro_unscope.hpp), so the view defines the few it needs +// under its own prefix; json_view.hpp undefines them all at its end +// (detail/view/macro_unscope.hpp). Configuration that json.hpp undefines is +// read from detail::abi_config instead. + +#if (defined(__cplusplus) && __cplusplus >= 201703L) || (defined(_MSVC_LANG) && _MSVC_LANG >= 201703L) + #define NLOHMANN_VIEW_HAS_CPP_17 1 +#else + #define NLOHMANN_VIEW_HAS_CPP_17 0 +#endif + +#if defined(__GNUC__) || defined(__clang__) + #define NLOHMANN_VIEW_LIKELY(x) __builtin_expect(!!(x), 1) + #define NLOHMANN_VIEW_UNLIKELY(x) __builtin_expect(!!(x), 0) + #define NLOHMANN_VIEW_ALWAYS_INLINE inline __attribute__((always_inline)) + #define NLOHMANN_VIEW_NOINLINE __attribute__((noinline)) +#elif defined(_MSC_VER) + #define NLOHMANN_VIEW_LIKELY(x) (x) + #define NLOHMANN_VIEW_UNLIKELY(x) (x) + #define NLOHMANN_VIEW_ALWAYS_INLINE __forceinline + #define NLOHMANN_VIEW_NOINLINE __declspec(noinline) +#else + #define NLOHMANN_VIEW_LIKELY(x) (x) + #define NLOHMANN_VIEW_UNLIKELY(x) (x) + #define NLOHMANN_VIEW_ALWAYS_INLINE inline + #define NLOHMANN_VIEW_NOINLINE +#endif + +#if defined(__GNUC__) || defined(__clang__) + #define NLOHMANN_VIEW_NODISCARD __attribute__((warn_unused_result)) +#elif defined(_MSC_VER) + #define NLOHMANN_VIEW_NODISCARD _Check_return_ +#else + #define NLOHMANN_VIEW_NODISCARD +#endif + +// exceptions as in json.hpp (JSON_NOEXCEPTION, JSON_THROW_USER) +#if (defined(__cpp_exceptions) || defined(__EXCEPTIONS) || defined(_CPPUNWIND)) && !defined(JSON_NOEXCEPTION) + #define NLOHMANN_VIEW_THROW(exception) throw exception +#else + #include + // (the exception is built first, so that the arguments of the throwing + // helpers count as used; the program ends anyway) + #define NLOHMANN_VIEW_THROW(exception) (static_cast(exception), std::abort()) +#endif +#if defined(JSON_THROW_USER) + #undef NLOHMANN_VIEW_THROW + #define NLOHMANN_VIEW_THROW JSON_THROW_USER +#endif + +// the parser stores a node's first word at once where the layout of `node` is +// known to be little-endian (MSVC targets are); elsewhere field by field +#if (defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__) || defined(_MSC_VER) + #define NLOHMANN_VIEW_LITTLE_ENDIAN 1 +#else + #define NLOHMANN_VIEW_LITTLE_ENDIAN 0 +#endif + +/// sixteen checks at fixed offsets 0..15 +#define NLOHMANN_VIEW_REPEAT16(X) X(0) X(1) X(2) X(3) X(4) X(5) X(6) X(7) X(8) X(9) X(10) X(11) X(12) X(13) X(14) X(15) + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t +#include // uint8_t, uint16_t, uint32_t, uint64_t +#include // memcpy + +// #include +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +// the node kinds are value_t values; the tests of is_container() and of the +// number kinds depend on this numbering +static_assert(static_cast(value_t::null) == 0 && static_cast(value_t::object) == 1 + && static_cast(value_t::array) == 2 && static_cast(value_t::string) == 3 + && static_cast(value_t::boolean) == 4 && static_cast(value_t::number_integer) == 5 + && static_cast(value_t::number_unsigned) == 6 && static_cast(value_t::number_float) == 7, + "the node format depends on the numbering of value_t"); + +/// node flags +struct node_flags +{ + static constexpr std::uint8_t escaped = 1; ///< string payload lives in the decode arena, not the source + static constexpr std::uint8_t storage = 3; ///< mask: where a string or number token lives (index into document_data::base) + static constexpr std::uint8_t is_true = 4; ///< boolean value +}; + +/// One entry of the flat index, in document order. An object's members are +/// stored as key node followed by the value's subtree. Integers keep their +/// converted 64-bit value in the len/next bytes (the node after a scalar is +/// always the next one, and the token length follows from `extra`). +struct node +{ + std::uint8_t kind; ///< value_t + std::uint8_t flags; ///< node_flags + std::uint16_t extra; ///< numbers: integer digits (low byte) and fraction digits (high byte), 255 = "many"; otherwise 0 + std::uint32_t off; ///< source offset (string content, number token, literal, bracket); arena offset if node_flags::escaped + std::uint32_t len; ///< string: decoded bytes; float: token bytes; array/object: element count + std::uint32_t next; ///< array/object: number of nodes of the subtree (its extent in the enclosing sequence) +}; +static_assert(sizeof(node) == 16, "node must stay 16 bytes"); + +NLOHMANN_VIEW_ALWAYS_INLINE bool is_container(const node& n) noexcept +{ + return static_cast(n.kind) - 1u <= 1u; +} + +/// the converted value of an integer node (stored in len/next) +NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t integer_bits(const node& n) noexcept +{ + std::uint64_t v = 0; + std::memcpy(&v, reinterpret_cast(&n) + 8, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + return v; +} + +NLOHMANN_VIEW_ALWAYS_INLINE void set_integer_bits(node& n, std::uint64_t v) noexcept +{ + std::memcpy(reinterpret_cast(&n) + 8, &v, 8); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) +} + +/// token length of a number node +NLOHMANN_VIEW_ALWAYS_INLINE std::uint32_t number_length(const node& n) noexcept +{ + return n.kind == static_cast(value_t::number_float) ? n.len + : (n.extra & 0xFFu) + (n.kind == static_cast(value_t::number_integer) ? 1u : 0u); +} + +/// estimated number of nodes for an input of `size` bytes (one node per ~12 +/// bytes covers typical documents without regrowth) +inline std::size_t estimate_nodes(std::size_t size) noexcept +{ + return (size / 12) + 16; +} + +/// estimated number of nodes for the input [src, src + size): pretty-printed +/// input (whitespace after the first byte) needs about a node per 12 bytes, +/// minified input up to one per 4 (yyjson tells the two apart the same way) +inline std::size_t estimate_nodes(const char* src, std::size_t size) noexcept +{ + return size >= 2 && (src[1] == ' ' || src[1] == '\n' || src[1] == '\r' || src[1] == '\t') ? estimate_nodes(size) : (size / 4) + 16; +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// storage of a parsed document; heap-allocated (header and an initial node +/// array in one block) so that views survive moves of the owning document +struct document_data +{ + const char* src = nullptr; + std::size_t size = 0; + node* tape = nullptr; + std::size_t tape_size = 0; + std::size_t tape_cap = 0; + node* inline_tape = nullptr; ///< node array allocated together with this header + std::size_t inline_cap = 0; + std::string arena{}; ///< decoded strings that contained escapes // NOLINT(readability-redundant-member-init) + std::string owned{}; ///< owned copy of the input, if any // NOLINT(readability-redundant-member-init) + std::array base = {{nullptr, nullptr, nullptr, nullptr}}; ///< string bases: source, arena (indexed by flags & node_flags::storage) + bool discarded = true; + + /// one allocation for the header and room for `nodes` nodes; large + /// documents get a separate node array instead (so it can be trimmed) + static document_data* create(std::size_t nodes) + { + nodes = nodes <= 256 ? nodes : 0; + void* mem = ::operator new (sizeof(document_data) + (nodes * sizeof(node))); + auto* d = new (mem) document_data(); // NOLINT(cppcoreguidelines-owning-memory): owned by the returned pointer, freed by deleter + // (aligned: sizeof is a multiple of the alignment; through void*, as GCC's -Wcast-align wants) + d->inline_tape = static_cast(static_cast(static_cast(mem) + sizeof(document_data))); // NOLINT(bugprone-casting-through-void) + d->inline_cap = nodes; + d->tape = d->inline_tape; + d->tape_cap = nodes; + return d; + } + + struct deleter + { + void operator()(document_data* d) const noexcept + { + d->~document_data(); + ::operator delete (d); + } + }; + + document_data() noexcept = default; + document_data(const document_data&) = delete; + document_data(document_data&&) = delete; + document_data& operator=(const document_data&) = delete; + document_data& operator=(document_data&&) = delete; + ~document_data() + { + release(); + } + + void release() noexcept + { + if (tape != inline_tape) + { + ::operator delete (tape); + } + tape = inline_tape; + tape_cap = inline_cap; + } + + /// make room for n nodes; keeps the first tape_size nodes + void reserve(std::size_t n) + { + if (n <= tape_cap) + { + return; + } + node* fresh = static_cast(::operator new (n * sizeof(node))); + if (tape_size != 0) + { + std::memcpy(fresh, tape, tape_size * sizeof(node)); + } + release(); + tape = fresh; + tape_cap = n; + } + + const char* str(const node& n) const noexcept + { + return base[n.flags & node_flags::storage] + n.off; + } + + /// the node after n's subtree (containers span `next` nodes, scalars one) + static NLOHMANN_VIEW_ALWAYS_INLINE const node* after(const node* n) noexcept + { + return n + (is_container(*n) ? n->next : 1u); + } + + /// first element (array) or first key (object) of a container + static NLOHMANN_VIEW_ALWAYS_INLINE const node* first_child(const node* n) noexcept + { + return n + 1; + } + + /// end of the elements of a container + static NLOHMANN_VIEW_ALWAYS_INLINE const node* child_end(const node* n) noexcept + { + return n + n->next; + } +}; + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +// #include + +// #include + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-FileCopyrightText: 2020 YaoYuan +// SPDX-License-Identifier: MIT + + + +#include // array +#include // size_t +#include // uint8_t, uint16_t, uint64_t +#include // memcpy + +// #include +// #include + + +// Scanning primitives of the view's parser. The unrolled checks at fixed +// offsets follow yyjson (https://github.com/ibireme/yyjson, MIT license): the +// loads do not depend on each other, so the CPU can run ahead. Words are read +// with read_eight_bytes(), so nothing here depends on the byte order. + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// 1 for bytes that may appear verbatim in a string: 0x20..0x7F except '"' and '\\' +inline const std::uint8_t* string_plain() noexcept +{ + static const std::array table = + { + { + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x00..0x1F + 1, 1, 0, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, // 0x20..0x3F ('"') + 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 1, 1, 1, // 0x40..0x5F ('\\') + 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, // 0x60..0x7F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0x80..0x9F + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xA0..0xBF + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xC0..0xDF + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, // 0xE0..0xFF + } + }; + return table.data(); +} + +NLOHMANN_VIEW_ALWAYS_INLINE bool is_digit(unsigned char c) noexcept +{ + return static_cast(c - '0') <= 9; +} + +/// two bytes as they are in memory (only compared with byte-symmetric patterns) +NLOHMANN_VIEW_ALWAYS_INLINE std::uint16_t load16(const unsigned char* p) noexcept +{ + std::uint16_t w = 0; + std::memcpy(&w, p, 2); + return w; +} + +/// Advance over plain string bytes and well-formed UTF-8. Stops at a quote, +/// a backslash, a control character, ill-formed UTF-8, or the end. The first +/// 16 bytes are checked one by one, so that the position advances by +/// constants in predicted branches (most strings are short); longer runs +/// continue eight bytes at a time. +NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* scan_string_run(const unsigned char* p, const unsigned char* e) noexcept +{ + const std::uint8_t* plain = string_plain(); + for (;;) + { + if (e - p >= 16) + { +#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(plain[p[i]] != 0)) {} else { p += (i); goto stop; } + NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) +#undef NLOHMANN_VIEW_STEP + p += 16; + while (e - p >= 8) + { + const std::uint64_t special = swar_string_special(read_eight_bytes(p)); + if (special != 0) + { + p += count_trailing_zeros(special) / 8; + goto stop; + } + p += 8; + } + continue; + } + while (p != e && plain[*p] != 0) + { + ++p; + } + if (p == e) + { + return p; + } +stop: + if (*p < 0x80) + { + return p; // quote, backslash, or control character + } + // non-ASCII: a run of well-formed sequences (the library's check, so + // that exactly what json::parse accepts is accepted) + do + { + const std::size_t n = validate_one_utf8(p, static_cast(e - p)); + if (n == 0) + { + return p; + } + p += n; + } + while (p != e && *p >= 0x80); + } +} + +/// advance over ASCII digits +NLOHMANN_VIEW_ALWAYS_INLINE const unsigned char* skip_digits(const unsigned char* p, const unsigned char* e) noexcept +{ + while (e - p >= 16) + { +#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(is_digit(p[i]))) {} else { return p + (i); } + NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) +#undef NLOHMANN_VIEW_STEP + p += 16; + } + while (p != e && is_digit(*p)) + { + ++p; + } + return p; +} + +/// powers of ten up to 10^19 as integers +inline std::uint64_t int_pow10(unsigned k) noexcept +{ + static const std::array table = + { + { + 1u, 10u, 100u, 1000u, 10000u, 100000u, 1000000u, 10000000u, 100000000u, 1000000000u, + 10000000000u, 100000000000u, 1000000000000u, 10000000000000u, 100000000000000u, 1000000000000000u, + 10000000000000000u, 100000000000000000u, 1000000000000000000u, 10000000000000000000u + } + }; + return table[k]; +} + +/// value of 0 < k < 8 digits at p in one step if [p, p + 8) lies below +/// limit, else one digit at a time (whole blocks of eight digits are read by +/// parse_upto19() directly) +NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t parse_upto8(const unsigned char* p, unsigned k, const unsigned char* limit) noexcept +{ + if (NLOHMANN_VIEW_LIKELY(limit - p >= 8)) + { + // move the k digits to the top and pad the vacated low bytes with '0' + const unsigned shift = 8 * (8 - k); + return parse_eight_digits((read_eight_bytes(p) << shift) | (0x3030303030303030u >> (8 * k))); + } + std::uint64_t v = 0; + for (unsigned i = 0; i < k; ++i) + { + v = (v * 10) + static_cast(p[i] - '0'); + } + return v; +} + +/// value of k <= 19 digits at p +NLOHMANN_VIEW_ALWAYS_INLINE std::uint64_t parse_upto19(const unsigned char* p, unsigned k, const unsigned char* limit) noexcept +{ + std::uint64_t w = 0; + while (k >= 8) + { + // (eight digits of the token: they lie below limit) + w = (w * 100000000u) + parse_eight_digits(read_eight_bytes(p)); + p += 8; + k -= 8; + } + if (k != 0) + { + w = (w * int_pow10(k)) + parse_upto8(p, k, limit); + } + return w; +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + + +// The view's parser: one pass over the input that emits the node index (see +// node.hpp). The table-driven decoding of \u escapes and the fast paths for +// ": " and indentation follow yyjson (https://github.com/ibireme/yyjson, MIT +// license). + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +enum class error_code : std::uint8_t +{ + none, + empty_input, + unexpected_value, + invalid_literal, + expected_key, + expected_colon, + expected_array_end, + expected_object_end, + trailing_characters, + number_after_minus, + number_after_dot, + number_after_exponent, + number_overflow, + string_missing_quote, + string_control_character, + string_utf8, + string_escape, + string_unicode_hex, + string_surrogate_high, + string_surrogate_low, + comment_start, + comment_unterminated, + input_too_large, +}; + +struct parse_failure +{ + error_code code = error_code::none; + std::size_t offset = 0; ///< byte offset of the offending character +}; + +/// FloatType: the number_float_t of the document, whose overflow parse() rejects +template +class builder +{ + public: + builder(document_data& d, const char* src, std::size_t size) noexcept + : doc(d) + , b(reinterpret_cast(src)) + , e(b + size) + {} + + /// returns false and fills `failure` on error + bool run() + { + cursor c(*this); + return c.run(); + } + + builder(const builder&) = delete; + builder& operator=(const builder&) = delete; + builder(builder&&) = delete; + builder& operator=(builder&&) = delete; + ~builder() = default; + + /// where and why the parse failed (after run() returned false) + const parse_failure& failure() const noexcept + { + return m_failure; + } + + private: + struct frame + { + std::uint32_t idx; + std::uint32_t count; + bool is_object; + }; + + document_data& doc; + const unsigned char* const b; + const unsigned char* const e; + parse_failure m_failure{}; + + // the open array/object is in the cursor; enclosing ones on a stack that is + // inline for the first 64 levels + frame shallow[64]; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays): not initialized on purpose; filled as containers open + std::vector deep{}; + + NLOHMANN_VIEW_NOINLINE bool fail(error_code c, const unsigned char* at) noexcept + { + m_failure.code = c; + m_failure.offset = static_cast(at - b); + doc.tape_size = 0; + return false; + } + + /// a failure (recorded by fail) as a comment() result + const unsigned char* fail_at(error_code c, const unsigned char* at) noexcept + { + fail(c, at); + return nullptr; + } + + /// a decoded string: p after its closing quote (nullptr: an error), and + /// its bytes in the arena + struct decoded + { + const unsigned char* p; + std::size_t start; + std::size_t len; + }; + + decoded failed(error_code c, const unsigned char* at) noexcept + { + fail(c, at); + return decoded{nullptr, 0, 0}; + } + + /// the comment at p (*p == '/'): the position after it, or nullptr on error + NLOHMANN_VIEW_NOINLINE const unsigned char* comment(const unsigned char* p) + { + if (e - p < 2) + { + ++p; + return fail_at(error_code::comment_start, p); + } + if (p[1] == '/') + { + p += 2; + // (as in parse(), a null byte is the end of the input, so it is + // left for the caller to see) + while (p != e && *p != '\n' && *p != '\r' && !(NulIsEnd && *p == 0)) + { + ++p; + } + return p; + } + if (p[1] == '*') + { + p += 2; + for (;;) + { + if (p == e || (NulIsEnd && *p == 0)) + { + return fail_at(error_code::comment_unterminated, p); + } + if (*p == '*' && p + 1 != e && p[1] == '/') + { + p += 2; + return p; + } + ++p; + } + } + ++p; + return fail_at(error_code::comment_start, p); + } + + /// the index is full (n nodes, parsed up to at): extrapolate the node + /// count from the nodes per input byte so far (with headroom, and at least + /// 1.5 times as many), so that dense inputs regrow once instead of + /// doubling repeatedly; returns the new node array + NLOHMANN_VIEW_NOINLINE node* grow(std::size_t n, const unsigned char* at) + { + const std::uint64_t done = static_cast(at - b) + 1; + const std::uint64_t guess = static_cast(n) * static_cast(e - b + 1) / done; + const std::uint64_t grown = guess + (guess / 4) + 64; // a variable: GCC calls a cast of the sum useless where std::uint64_t is std::size_t + doc.tape_size = n; + doc.reserve((std::max)(static_cast(grown), n + (n / 2) + 64)); + return doc.tape; + } + + /// four hex digits at p as a code unit (p moves past them), or -1 (p at + /// the first bad digit); one table lookup per digit and a single check, + /// the four hex digits of a unicode escape (the library's table, after + /// yyjson's read_hex_u16), or -1 + NLOHMANN_VIEW_ALWAYS_INLINE int hex4(const unsigned char*& p) noexcept + { + if (NLOHMANN_VIEW_LIKELY(e - p >= 4)) + { + const int cp = hex_codepoint(p); + if (NLOHMANN_VIEW_LIKELY(cp >= 0)) + { + p += 4; + return cp; + } + } + p = hex4_error(p); + return -1; + } + + /// hex4() failed: the first bad digit (none: p) + NLOHMANN_VIEW_NOINLINE const unsigned char* hex4_error(const unsigned char* p) noexcept + { + if (e - p >= 4) + { + while (is_hex(*p)) + { + ++p; + } + } + return p; + } + + /// escapes present (or an error) in the string at s, scanned up to p: + /// decode into the arena + NLOHMANN_VIEW_NOINLINE decoded slow_string(const unsigned char* s, const unsigned char* p) + { + // single-character escapes; 0: invalid (and 'u', handled separately) + static const std::array simple_escape = + { + { + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, '"', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '/', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, + 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, '\\', 0, 0, 0, + 0, 0, '\b', 0, 0, 0, '\f', 0, 0, 0, 0, 0, 0, 0, '\n', 0, 0, 0, '\r', 0, '\t', 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0 + } + }; + const std::size_t start = arena_used(); + arena_run(s, static_cast(p - s)); + for (;;) + { + if (p == e) + { + return failed(error_code::string_missing_quote, p); + } + const unsigned char c = *p; + if (c == '"') + { + ++p; + return decoded{p, start, arena_used() - start}; + } + if (c != '\\') + { + // (a NUL before the end of the input is a control character, as + // for json::parse, also where a NUL ends the input between values) + return failed(c < 0x20 ? error_code::string_control_character : error_code::string_utf8, p); + } + ++p; + if (p == e) + { + return failed(error_code::string_missing_quote, p); + } + const unsigned char d = *p++; + arena_ensure(4); + if (d == 'u') + { + int cp = hex4(p); + if (NLOHMANN_VIEW_UNLIKELY(cp < 0)) + { + return failed(error_code::string_unicode_hex, p); + } + if (NLOHMANN_VIEW_UNLIKELY((cp & 0xF800) == 0xD800)) // a surrogate + { + if (cp >= 0xDC00) + { + return failed(error_code::string_surrogate_low, p); + } + if (e - p < 2 || p[0] != '\\' || p[1] != 'u') + { + return failed(error_code::string_surrogate_high, p); + } + p += 2; + const int lo = hex4(p); + if (lo < 0) + { + return failed(error_code::string_unicode_hex, p); + } + if (lo < 0xDC00 || lo > 0xDFFF) + { + return failed(error_code::string_surrogate_high, p); + } + cp = 0x10000 + ((cp - 0xD800) << 10) + (lo - 0xDC00); + } + aw = put_utf8(aw, cp); + } + else if (NLOHMANN_VIEW_LIKELY(d < 128 && simple_escape[d] != 0)) + { + *aw++ = simple_escape[d]; + } + else + { + --p; + return failed(error_code::string_escape, p); + } + if (p != e && *p == '\\') + { + continue; // consecutive escapes ("\u00e4\u00f6"): no run in between + } + const unsigned char* const r = p; + p = scan_string_run(p, e); + arena_run(r, static_cast(p - r)); + } + } + + /// does the float token [s, p) overflow FloatType? (parse() rejects it) + NLOHMANN_VIEW_NOINLINE static bool float_overflows(const unsigned char* s, const unsigned char* p) + { + const auto* const first = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const auto* const last = reinterpret_cast(p); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + const char* const dot = std::find(first, last, '.'); + const char* const exponent = std::find_if(first, last, [](char c) + { + return c == 'e' || c == 'E'; + }); + const auto v = convert_float(first, last, dot == last ? std::string::npos : static_cast(dot - first), + static_cast(exponent - first)); + return v > (std::numeric_limits::max)() || v < -(std::numeric_limits::max)(); + } + + /// does the magnitude digits [d, d + n) exceed the given limit (same length)? + static bool digits_exceed(const unsigned char* d, const char* limit, std::size_t n) noexcept + { + return std::memcmp(d, limit, n) > 0; + } + + static bool is_hex(unsigned char c) noexcept + { + return (c >= '0' && c <= '9') || (c >= 'A' && c <= 'F') || (c >= 'a' && c <= 'f'); + } + + // decode arena: the std::string in the document, written through a raw + // pointer (resized ahead in large steps; trimmed when parsing succeeds) + char* aw = nullptr; + char* aend = nullptr; + + NLOHMANN_VIEW_ALWAYS_INLINE std::size_t arena_used() const noexcept + { + return aw != nullptr ? static_cast(aw - doc.arena.data()) : 0; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void arena_ensure(std::size_t n) + { + if (NLOHMANN_VIEW_UNLIKELY(static_cast(aend - aw) < n)) + { + arena_grow(n); + } + } + + NLOHMANN_VIEW_NOINLINE void arena_grow(std::size_t n) + { + const std::size_t used = arena_used(); + doc.arena.resize((std::max)(doc.arena.size() * 2, used + n + 256)); + aw = &doc.arena[0] + used; // NOLINT(readability-container-data-pointer): data() is const before C++17 + aend = &doc.arena[0] + doc.arena.size(); // NOLINT(readability-container-data-pointer) + } + + /// append the run [r, r + n) to the arena; short runs as one fixed-size + /// 16-byte move when both sides have the room (no library call) + NLOHMANN_VIEW_ALWAYS_INLINE void arena_run(const unsigned char* r, std::size_t n) + { + arena_ensure(n + 16); + if (n <= 16 && e - r >= 16) + { + std::memcpy(aw, r, 16); + } + else + { + std::memcpy(aw, r, n); + } + aw += n; + } + + /// UTF-8 encoding of cp at w (room for 4 bytes) + static char* put_utf8(char* w, int cp) noexcept + { + if (cp < 0x80) + { + *w++ = static_cast(cp); + } + else if (cp < 0x800) + { + *w++ = static_cast(0xC0 | (cp >> 6)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + else if (cp < 0x10000) + { + *w++ = static_cast(0xE0 | (cp >> 12)); + *w++ = static_cast(0x80 | ((cp >> 6) & 0x3F)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + else + { + *w++ = static_cast(0xF0 | (cp >> 18)); + *w++ = static_cast(0x80 | ((cp >> 12) & 0x3F)); + *w++ = static_cast(0x80 | ((cp >> 6) & 0x3F)); + *w++ = static_cast(0x80 | (cp & 0x3F)); + } + return w; + } + + /// a compile-time option as a runtime condition: testing the template + /// argument directly makes a condition like `TrailingCommas && c == ']'` + /// constant when the option is off, which MSVC reports as C4127 + static NLOHMANN_VIEW_ALWAYS_INLINE bool enabled(bool option) noexcept + { + return option; + } + + /// The parse state and the parser proper. The cursor is a local object of + /// run() whose address never escapes (everything it calls out of line is a + /// member of the builder and gets the positions it needs), so that the + /// compiler keeps the state in registers instead of reloading it from + /// memory after every node store and call. + struct cursor + { + explicit cursor(builder& owner) noexcept + : cold(owner) + , b(owner.b) + , p(owner.b) + , e(owner.e) + {} + + builder& cold; ///< out-of-line helpers and state that needs no registers + const unsigned char* const b; + const unsigned char* p; + const unsigned char* const e; + node* base = nullptr; + node* out = nullptr; + node* cap = nullptr; + + // the open array/object + std::uint32_t cur_idx = 0; + std::uint32_t cur_count = 0; + bool cur_is_object = false; + std::size_t depth = 0; + + NLOHMANN_VIEW_ALWAYS_INLINE bool run() + { + cold.doc.reserve(estimate_nodes(reinterpret_cast(b), static_cast(e - b))); + base = cold.doc.tape; + out = base; + cap = base + cold.doc.tape_cap; + + if (e - p >= 3 && p[0] == 0xEF && p[1] == 0xBB && p[2] == 0xBF) + { + p += 3; // byte order mark + } + if (!ws()) + { + return false; + } + if (p == e || (NulIsEnd && *p == 0)) + { + return fail(error_code::empty_input); + } + + // root value + switch (cur()) + { + case '{': + open(value_t::object); + ++p; + goto obj_first; + case '[': + open(value_t::array); + ++p; + goto arr_first; + default: + if (!scalar()) + { + return false; + } + goto root_done; + } + + // value dispatch, expanded once for array elements and once for member + // values: each jump has its own history (arrays tend to hold one kind of + // value), and the continuation needs no branch on the container kind +#define NLOHMANN_VIEW_VALUE(NEXT) \ + switch (cur()) \ + { \ + case '"': \ + if (NLOHMANN_VIEW_UNLIKELY(!string())) { return false; } \ + goto NEXT; \ + case '{': \ + open(value_t::object); \ + ++p; \ + goto obj_first; \ + case '[': \ + open(value_t::array); \ + ++p; \ + goto arr_first; \ + case '-': \ + if (NLOHMANN_VIEW_UNLIKELY(!number())) { return false; } \ + goto NEXT; \ + case '0': case '1': case '2': case '3': \ + case '4': case '5': case '6': case '7': case '8': case '9': \ + if (NLOHMANN_VIEW_UNLIKELY(!number())) { return false; } \ + goto NEXT; \ + case 't': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal("true", 4, value_t::boolean, node_flags::is_true))) { return false; } \ + goto NEXT; \ + case 'f': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal_false())) { return false; } \ + goto NEXT; \ + case 'n': \ + if (NLOHMANN_VIEW_UNLIKELY(!literal("null", 4, value_t::null, 0))) { return false; } \ + goto NEXT; \ + default: \ + return fail(error_code::unexpected_value); \ + } + +arr_first: + if (!ws()) + { + return false; + } + if (cur() == ']') + { + ++p; + goto close_container; + } +value: + NLOHMANN_VIEW_VALUE(arr_next) +arr_next: + ++cur_count; + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ',')) + { + ++p; + if (!ws()) + { + return false; + } + if (enabled(TrailingCommas) && cur() == ']') + { + ++p; + goto close_container; + } + goto value; + } + if (cur() == ']') + { + ++p; + goto close_container; + } + return fail(error_code::expected_array_end); + +obj_first: + if (!ws()) + { + return false; + } + if (cur() == '}') + { + ++p; + goto close_container; + } +obj_key: + if (NLOHMANN_VIEW_UNLIKELY(cur() != '"')) + { + return fail(error_code::expected_key); + } + if (NLOHMANN_VIEW_UNLIKELY(!string())) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ':' && (Sentinel || e - p >= 2) && p[1] == ' ')) + { + p += 2; // ": " (pretty-printed input; a fast path of yyjson) + } + else + { + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_UNLIKELY(cur() != ':')) + { + return fail(error_code::expected_colon); + } + ++p; + } + if (!ws()) + { + return false; + } + NLOHMANN_VIEW_VALUE(obj_next) +obj_next: + ++cur_count; + if (!ws()) + { + return false; + } + if (NLOHMANN_VIEW_LIKELY(cur() == ',')) + { + ++p; + if (!ws()) + { + return false; + } + if (enabled(TrailingCommas) && cur() == '}') + { + ++p; + goto close_container; + } + goto obj_key; + } + if (cur() == '}') + { + ++p; + goto close_container; + } + return fail(error_code::expected_object_end); + +#undef NLOHMANN_VIEW_VALUE + +close_container: + close(); + if (NLOHMANN_VIEW_UNLIKELY(depth == 0)) + { + goto root_done; + } + if (cur_is_object) + { + goto obj_next; + } + goto arr_next; + +root_done: + if (!ws()) + { + return false; + } + if (p != e && !(NulIsEnd && *p == 0)) + { + return fail(error_code::trailing_characters); + } + cold.doc.tape_size = static_cast(out - base); + cold.doc.arena.resize(cold.arena_used()); + return true; + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool fail(error_code c) noexcept + { + return cold.fail(c, p); + } + + /// the current byte, or 0 at the end. With a NUL-terminated input + /// (Sentinel) the terminator is read instead of checking the bounds; a 0 + /// never matches a JSON token, so the error paths tell the end apart. + NLOHMANN_VIEW_ALWAYS_INLINE unsigned char cur() const noexcept + { + if (Sentinel) + { + return *p; + } + return p != e ? *p : 0; + } + + /// root scalar + NLOHMANN_VIEW_ALWAYS_INLINE bool scalar() + { + switch (cur()) + { + case '"': + return string(); + case 't': + return literal("true", 4, value_t::boolean, node_flags::is_true); + case 'f': + return literal_false(); + case 'n': + return literal("null", 4, value_t::null, 0); + case '-': + return number(); + case '0': + case '1': + case '2': + case '3': + case '4': + case '5': + case '6': + case '7': + case '8': + case '9': + return number(); + default: + return fail(error_code::unexpected_value); + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool literal_false() + { + return literal("false", 5, value_t::boolean, 0); + } + + /// skip whitespace (and comments); false on a malformed comment + NLOHMANN_VIEW_ALWAYS_INLINE bool ws() + { + const unsigned char c = cur(); + if (NLOHMANN_VIEW_LIKELY(c > ' ' && (!Comments || c != '/'))) + { + return true; // no whitespace: the common case in minified input + } + return ws_slow(); + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool ws_slow() + { + for (;;) + { + if (cur() == ' ' && (Sentinel || e - p >= 2) && p[1] > ' ' && (!Comments || p[1] != '/')) + { + ++p; // single space, e.g. after ':' or ',' + return true; + } + if (cur() == '\n' || cur() == '\r') + { + // (a branch, not an add of the comparison: p must not + // wait for the byte after the line break) + if (NLOHMANN_VIEW_UNLIKELY(cur() == '\r') && (Sentinel || e - p >= 2) && p[1] == '\n') + { + p += 2; + } + else + { + ++p; + } + // indentation: two spaces per step, fixed offsets (after yyjson) + while (e - p >= 32) + { +#define NLOHMANN_VIEW_STEP(i) if (NLOHMANN_VIEW_LIKELY(load16(p + (std::ptrdiff_t{2} * (i))) == 0x2020)) {} else { p += std::ptrdiff_t{2} * (i); goto indent_done; } + NLOHMANN_VIEW_REPEAT16(NLOHMANN_VIEW_STEP) +#undef NLOHMANN_VIEW_STEP + p += 32; + } +indent_done: + ; + } + for (unsigned char c = cur(); c == ' ' || c == '\n' || c == '\r' || c == '\t'; c = cur()) + { + ++p; + } + if (enabled(Comments) && cur() == '/') + { + const unsigned char* const q = cold.comment(p); + if (q == nullptr) + { + return false; + } + p = q; + continue; + } + return true; + } + } + + /// append a node: (kind, flags, extra, off) and the second word (len, or + /// an integer's value); two stores on little-endian targets + NLOHMANN_VIEW_ALWAYS_INLINE node* emit(value_t k, std::uint8_t flags, std::uint16_t extra, std::size_t off, std::uint64_t second) + { + if (NLOHMANN_VIEW_UNLIKELY(out == cap)) + { + const auto n = static_cast(out - base); + base = cold.grow(n, p); + out = base + n; + cap = base + cold.doc.tape_cap; + } + node* n = out++; +#if NLOHMANN_VIEW_LITTLE_ENDIAN + const std::uint64_t first = static_cast(k) | (static_cast(flags) << 8) + | (static_cast(extra) << 16) | (static_cast(off) << 32); + std::memcpy(reinterpret_cast(n), &first, 8); + std::memcpy(reinterpret_cast(n) + 8, &second, 8); +#else + n->kind = static_cast(k); + n->flags = flags; + n->extra = extra; + n->off = static_cast(off); + set_integer_bits(*n, second); +#endif + return n; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void open(value_t k) + { + const auto idx = static_cast(emit(k, 0, 0, static_cast(p - b), 0) - base); + if (depth != 0) + { + const frame f = {cur_idx, cur_count, cur_is_object}; + if (NLOHMANN_VIEW_LIKELY(depth <= 64)) + { + cold.shallow[depth - 1] = f; + } + else + { + cold.deep.push_back(f); + } + } + ++depth; + cur_idx = idx; + cur_count = 0; + cur_is_object = k == value_t::object; + } + + NLOHMANN_VIEW_ALWAYS_INLINE void close() + { + node& n = base[cur_idx]; + n.len = cur_count; + n.next = static_cast(out - base) - cur_idx; + if (--depth != 0) + { + frame f{}; + if (NLOHMANN_VIEW_LIKELY(depth <= 64)) + { + f = cold.shallow[depth - 1]; + } + else + { + f = cold.deep.back(); + cold.deep.pop_back(); + } + cur_idx = f.idx; + cur_count = f.count; + cur_is_object = f.is_object; + } + } + + NLOHMANN_VIEW_ALWAYS_INLINE bool literal(const char* text, std::size_t n, value_t k, std::uint8_t flags) + { + if (NLOHMANN_VIEW_UNLIKELY(e - p < static_cast(n) || std::memcmp(p, text, n) != 0)) + { + return fail(error_code::invalid_literal); + } + emit(k, flags, 0, static_cast(p - b), n); + p += n; + return true; + } + + /// a number at p; the sign is known from the dispatch (so that p does + /// not have to wait for the first byte) + template + NLOHMANN_VIEW_ALWAYS_INLINE bool number() + { + const unsigned char* const s = p; + if (negative) + { + ++p; + } + const unsigned char* const int_start = p; + if (p != e && *p == '0') + { + ++p; + } + else if (NLOHMANN_VIEW_LIKELY(p != e && *p >= '1' && *p <= '9')) + { + p = skip_digits(p + 1, e); + } + else + { + return fail(error_code::number_after_minus); + } + const auto int_digits = static_cast(p - int_start); + std::size_t frac_digits = 0; + bool is_float = false; + if (p != e && *p == '.') + { + ++p; + const unsigned char* const f0 = p; + p = skip_digits(p, e); + if (NLOHMANN_VIEW_UNLIKELY(p == f0)) + { + return fail(error_code::number_after_dot); + } + frac_digits = static_cast(p - f0); + is_float = true; + } + std::int64_t exponent = 0; + if (p != e && (*p | 0x20) == 'e') + { + ++p; + bool exp_negative = false; + if (p != e && (*p == '+' || *p == '-')) + { + exp_negative = *p == '-'; + ++p; + } + if (NLOHMANN_VIEW_UNLIKELY(p == e || !is_digit(*p))) + { + return fail(error_code::number_after_exponent); + } + while (p != e && is_digit(*p)) + { + if (exponent < 100000) + { + exponent = (exponent * 10) + (*p - '0'); + } + ++p; + } + if (exp_negative) + { + exponent = -exponent; + } + is_float = true; + } + + value_t kind = value_t::number_float; + if (!is_float) + { + kind = negative ? value_t::number_integer : value_t::number_unsigned; + } + if (!is_float) + { + // integers that do not fit become floats, as in parse() + if (NLOHMANN_VIEW_UNLIKELY(int_digits >= 19)) + { + if (negative) + { + if (int_digits > 19 || (int_digits == 19 && digits_exceed(int_start, "9223372036854775808", 19))) + { + kind = value_t::number_float; + } + } + else if (int_digits > 20 || (int_digits == 20 && digits_exceed(int_start, "18446744073709551615", 20))) + { + kind = value_t::number_float; + } + } + } + // parse() rejects floats that overflow; only numbers whose magnitude + // could reach the largest FloatType (1e308 for double, 1e38 for + // float) need the conversion + if (NLOHMANN_VIEW_UNLIKELY(static_cast(int_digits) + exponent > std::numeric_limits::max_exponent10 - 8 && kind == value_t::number_float)) + { + if (builder::float_overflows(s, p)) + { + p = s; + return fail(error_code::number_overflow); + } + } + const auto layout = static_cast((int_digits < 255 ? int_digits : 255) | ((frac_digits < 255 ? frac_digits : 255) << 8)); + auto second = static_cast(p - s); + if (kind != value_t::number_float) + { + // integers are converted now, while their digits are in cache + const std::uint64_t m = int_digits <= 19 ? parse_upto19(int_start, static_cast(int_digits), e) + : (parse_upto19(int_start, 19, e) * 10) + static_cast(int_start[19] - '0'); + second = negative ? 0 - m : m; + } + emit(kind, 0, layout, static_cast(s - b), second); + return true; + } + + /// a string (value or key) at p + NLOHMANN_VIEW_ALWAYS_INLINE bool string() + { + ++p; // opening quote + const unsigned char* const s = p; + p = scan_string_run(p, e); + if (NLOHMANN_VIEW_LIKELY(p != e && *p == '"')) + { + emit(value_t::string, 0, 0, static_cast(s - b), static_cast(p - s)); + ++p; + return true; + } + const decoded r = cold.slow_string(s, p); + if (r.p == nullptr) + { + return false; + } + p = r.p; + emit(value_t::string, node_flags::escaped, 0, r.start, r.len); + return true; + } + }; +}; + +/// run the builder with compile-time options +template +inline bool build_with(document_data& d, const char* src, std::size_t size, bool sentinel, parse_failure& failure) +{ + if (sentinel) + { + builder bld(d, src, size); + const bool ok = bld.run(); + failure = bld.failure(); + return ok; + } + builder bld(d, src, size); + const bool ok = bld.run(); + failure = bld.failure(); + return ok; +} + +/// sentinel: src[size] is readable and 0 (e.g. std::string); FloatType: the +/// number_float_t of the document +template +inline bool build(document_data& d, const char* src, std::size_t size, bool comments, bool trailing_commas, bool sentinel, parse_failure& failure) +{ + if (comments) + { + return trailing_commas ? build_with(d, src, size, sentinel, failure) + : build_with(d, src, size, sentinel, failure); + } + return trailing_commas ? build_with(d, src, size, sentinel, failure) + : build_with(d, src, size, sentinel, failure); +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +// #include + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // min +#include // size_t +#include // string + +// #include +// #include + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +// Exceptions are thrown out of line, so that the accessors that may throw stay +// small enough to be inlined. + +[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_type_error(int id, const char* prefix, const char* type) +{ + NLOHMANN_VIEW_THROW(type_error::create(id, concat(prefix, type), nullptr)); +} + +[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_out_of_range(int id, const std::string& msg) +{ + NLOHMANN_VIEW_THROW(out_of_range::create(id, msg, nullptr)); +} + +[[noreturn]] NLOHMANN_VIEW_NOINLINE inline void throw_invalid_iterator(int id, const char* msg) +{ + NLOHMANN_VIEW_THROW(invalid_iterator::create(id, msg, nullptr)); +} + +/*! +@brief throw the exception BasicJsonType::parse would throw for this input + +The view accepts exactly the inputs parse() accepts, so on a failure the +library parser is run on the same bytes: it throws the exception parse() would +throw, with the same message, position, and "last read" token. The error path +is cold, so this costs nothing on valid input. Should parse() accept the input +nevertheless (a bug), the view's own failure is reported. +*/ +template +[[noreturn]] NLOHMANN_VIEW_NOINLINE void throw_parse_failure(const parse_failure& f, const char* src, std::size_t size, + bool ignore_comments, bool ignore_trailing_commas) +{ + if (f.code == error_code::input_too_large) + { + // LCOV_EXCL_START (4 GiB) + NLOHMANN_VIEW_THROW(out_of_range::create(416, "input of 4 GiB or more is not supported by json_document", nullptr)); + // LCOV_EXCL_STOP + } + const BasicJsonType accepted = BasicJsonType::parse(src, src + size, nullptr, true, ignore_comments, ignore_trailing_commas); + // LCOV_EXCL_START (only if parse() accepts what the view rejects: a bug) + static_cast(accepted); + + position_t pos; + const std::size_t off = (std::min)(f.offset, size); + pos.chars_read_total = off + 1; + std::size_t line_start = 0; + for (std::size_t i = 0; i < off; ++i) + { + if (src[i] == '\n') + { + ++pos.lines_read; + line_start = i + 1; + } + } + pos.chars_read_current_line = off + 1 - line_start; + NLOHMANN_VIEW_THROW(parse_error::create(101, pos, "syntax error while parsing value", nullptr)); + // LCOV_EXCL_STOP +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // basic_string, char_traits, string +#include // decay, integral_constant, is_array, is_lvalue_reference, is_pointer, is_same, remove_reference +#include // forward + +// #include +// #include + + +#if NLOHMANN_VIEW_HAS_CPP_17 + #include // string_view +#endif + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/// how a document takes its input +enum class input_kind +{ + move_string, ///< rvalue std::string: owned without a copy + c_string, ///< const char* (NUL-terminated): borrowed + char_array, ///< char array (e.g. a string literal): borrowed + borrow_range, ///< lvalue contiguous byte container, or std::string_view: borrowed + copy_range, ///< rvalue contiguous byte container: copied + adapter, ///< anything else parse() accepts (streams, wide strings, ...): read into a buffer +}; + +template +struct classify_input +{ + using R = typename std::remove_reference::type; + using D = typename std::decay::type; + static constexpr bool is_rvalue = !std::is_lvalue_reference::value; + static constexpr bool is_bytes = is_contiguous_byte_container::value; +#if NLOHMANN_VIEW_HAS_CPP_17 + static constexpr bool is_string_view = std::is_same::value; +#else + static constexpr bool is_string_view = false; +#endif + // NOLINTBEGIN(readability-avoid-nested-conditional-operator): a constant expression of C++11 + static constexpr input_kind value = + std::is_array::value ? input_kind::char_array + : std::is_pointer::value ? input_kind::c_string + : (is_rvalue && std::is_same::value) ? input_kind::move_string + : (is_bytes && (!is_rvalue || is_string_view)) ? input_kind::borrow_range + : is_bytes ? input_kind::copy_range + : input_kind::adapter; + // NOLINTEND(readability-avoid-nested-conditional-operator) +}; + +/// std::basic_string guarantees a NUL at data()[size()] (the parser's sentinel) +template +struct is_std_string : std::false_type {}; + +template +struct is_std_string> : std::true_type {}; + +/// drain a json input adapter (UTF-16/32 inputs arrive as UTF-8) +template +std::string collect_adapter(Adapter ia) +{ + std::string buf; + for (;;) + { + const auto ch = ia.get_character(); + if (ch == std::char_traits::eof()) + { + break; + } + buf.push_back(static_cast(ch)); + } + return buf; +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +// #include + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // int64_t, uint8_t +#include // string +#include // vector + +// #include +// #include + +// #include + +// #include + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // size_t +#include // string + +// #include +// #include + +// #include + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/*! +@brief the value of the float token of a node, as parse() converts it + +Uses the lexer's conversion (detail::convert_float), so that the values are +bit-identical to parse(): float and double are converted without allocation +and independent of the locale. The digit layout recorded while parsing locates +the decimal point and the exponent without scanning the token. +*/ +template +NLOHMANN_VIEW_NOINLINE FloatType float_value(const char* first, const node& n) +{ + const char* const last = first + n.len; + const std::size_t neg = first[0] == '-' ? 1 : 0; + const std::size_t int_digits = n.extra & 0xFFu; + const std::size_t frac_digits = n.extra >> 8u; + std::size_t dot = std::string::npos; + std::size_t mantissa_end = n.len; + if (int_digits != 255 && frac_digits != 255) + { + dot = frac_digits != 0 ? neg + int_digits : std::string::npos; + mantissa_end = neg + int_digits + (frac_digits != 0 ? 1 + frac_digits : 0); + } + else + { + // more digits than the layout records: locate them + for (std::size_t i = 0; i < n.len; ++i) + { + if (first[i] == '.') + { + dot = i; + } + else if (first[i] == 'e' || first[i] == 'E') + { + mantissa_end = i; + break; + } + } + } + return convert_float(first, last, dot, mantissa_end); +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +/*! +@brief the basic_json value of the subtree at n + +The subtree is replayed into the SAX handler that parse() uses to build its +values, so the result is the value parse() would produce: duplicate keys keep +the last value, and with JSON_DIAGNOSTICS the parent pointers are set. It is +iterative, so the nesting depth is limited by memory only, as for parse(). +Without a lexer the handler records no source positions +(JSON_DIAGNOSTIC_POSITIONS). +*/ +template +BasicJsonType materialize(const document_data& d, const node* n) +{ + using string_t = typename BasicJsonType::string_t; + using sax_t = json_sax_dom_parser>; + + BasicJsonType result; + sax_t sax(result, true); + const string_t no_token{}; + // the ends of the open containers, and whether they are objects + std::vector> open; + for (;;) + { + switch (static_cast(n->kind)) + { + case value_t::object: + case value_t::array: + { + const bool object = n->kind == static_cast(value_t::object); + if (object) + { + sax.start_object(n->len); + } + else + { + sax.start_array(n->len); + } + open.emplace_back(document_data::child_end(n), object); + n = document_data::first_child(n); + break; + } + case value_t::string: + { + string_t s(d.str(*n), n->len); + sax.string(s); + ++n; + break; + } + case value_t::number_integer: + sax.number_integer(static_cast(static_cast(integer_bits(*n)))); + ++n; + break; + case value_t::number_unsigned: + sax.number_unsigned(static_cast(integer_bits(*n))); + ++n; + break; + case value_t::number_float: + sax.number_float(float_value(d.str(*n), *n), no_token); + ++n; + break; + case value_t::boolean: + sax.boolean((n->flags & node_flags::is_true) != 0); + ++n; + break; + case value_t::null: + case value_t::binary: + case value_t::discarded: + default: + sax.null(); + ++n; + break; + } + for (;;) + { + if (open.empty()) + { + return result; + } + if (n != open.back().first) + { + break; + } + if (open.back().second) + { + sax.end_object(); + } + else + { + sax.end_array(); + } + open.pop_back(); + } + if (open.back().second) + { + // the key of the next member + string_t key(d.str(*n), n->len); + sax.key(key); + ++n; + } + } +} + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + +// #include + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +#include // min +#include // size_t +#include // memcmp, strlen +#include // basic_string + +// #include +// #include + + +#if NLOHMANN_VIEW_HAS_CPP_17 + #include // string_view +#endif +#ifndef JSON_NO_IO + #include // ostream +#endif + +NLOHMANN_JSON_NAMESPACE_BEGIN +namespace detail +{ +namespace view +{ + +#if NLOHMANN_VIEW_HAS_CPP_17 +using string_ref = std::string_view; +#else +/// minimal C++11 stand-in for std::string_view +class string_ref +{ + public: + using size_type = std::size_t; + using const_iterator = const char*; + + string_ref() noexcept = default; + // s must be null-terminated, as for std::string_view(const char*) + // flawfinder: ignore + string_ref(const char* s) : m_data(s), m_size(std::strlen(s)) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + string_ref(const char* s, std::size_t n) noexcept : m_data(s), m_size(n) {} + template + string_ref(const std::basic_string& s) noexcept : m_data(s.data()), m_size(s.size()) {} // NOLINT(google-explicit-constructor,hicpp-explicit-conversions) + + const char* data() const noexcept + { + return m_data; + } + std::size_t size() const noexcept + { + return m_size; + } + std::size_t length() const noexcept + { + return m_size; + } + bool empty() const noexcept + { + return m_size == 0; + } + const char* begin() const noexcept + { + return m_data; + } + const char* end() const noexcept + { + return m_data + m_size; + } + char operator[](std::size_t i) const noexcept + { + return m_data[i]; + } + + template + explicit operator std::basic_string() const + { + return std::basic_string(m_data, m_size); + } + + friend bool operator==(string_ref a, string_ref b) noexcept + { + return a.m_size == b.m_size && (a.m_size == 0 || std::memcmp(a.m_data, b.m_data, a.m_size) == 0); + } + friend bool operator!=(string_ref a, string_ref b) noexcept + { + return !(a == b); + } + friend bool operator<(string_ref a, string_ref b) noexcept + { + const int c = std::memcmp(a.m_data, b.m_data, (std::min)(a.m_size, b.m_size)); + return c != 0 ? c < 0 : a.m_size < b.m_size; + } +#ifndef JSON_NO_IO + friend std::ostream& operator<<(std::ostream& o, string_ref s) + { + return o.write(s.m_data, static_cast(s.m_size)); + } +#endif + + private: + const char* m_data = ""; + std::size_t m_size = 0; +}; +#endif + +} // namespace view +} // namespace detail +NLOHMANN_JSON_NAMESPACE_END + + +NLOHMANN_JSON_NAMESPACE_BEGIN + +template +class basic_json_document; + +/*! +@brief read-only handle to one value of a basic_json_document + +Trivially copyable (two pointers). Valid as long as the document is alive and +has not been re-parsed, and as long as a borrowed source text is alive. +*/ +template +class basic_json_view +{ + using node = detail::view::node; + using document_data = detail::view::document_data; + + public: + using value_t = detail::value_t; + using string_t = typename BasicJsonType::string_t; + using number_integer_t = typename BasicJsonType::number_integer_t; + using number_unsigned_t = typename BasicJsonType::number_unsigned_t; + using number_float_t = typename BasicJsonType::number_float_t; + using json_pointer = typename BasicJsonType::json_pointer; + using size_type = std::size_t; + /// std::string_view from C++17 on + using string_view_t = detail::view::string_ref; + + /// an invalid view (type() == value_t::discarded) + basic_json_view() noexcept = default; + + ////////// + // type // + ////////// + + NLOHMANN_VIEW_ALWAYS_INLINE value_t type() const noexcept + { + return m_node != nullptr ? static_cast(m_node->kind) : value_t::discarded; + } + + bool is_null() const noexcept + { + return type() == value_t::null; + } + + bool is_boolean() const noexcept + { + return type() == value_t::boolean; + } + + bool is_number() const noexcept + { + return is_number_integer() || is_number_float(); + } + + bool is_number_integer() const noexcept + { + return type() == value_t::number_integer || type() == value_t::number_unsigned; + } + + bool is_number_unsigned() const noexcept + { + return type() == value_t::number_unsigned; + } + + bool is_number_float() const noexcept + { + return type() == value_t::number_float; + } + + bool is_string() const noexcept + { + return type() == value_t::string; + } + + bool is_array() const noexcept + { + return type() == value_t::array; + } + + bool is_object() const noexcept + { + return type() == value_t::object; + } + + /// always false: JSON text has no binary values + bool is_binary() const noexcept + { + return false; + } + + bool is_primitive() const noexcept + { + return is_null() || is_string() || is_boolean() || is_number(); + } + + bool is_structured() const noexcept + { + return is_array() || is_object(); + } + + /// the root of a failed parse with allow_exceptions == false, or a + /// default-constructed view + bool is_discarded() const noexcept + { + return type() == value_t::discarded; + } + + /// false for discarded views + explicit operator bool() const noexcept + { + return m_node != nullptr; + } + + ////////////// + // capacity // + ////////////// + + /// the number of elements (arrays, objects), 0 for null and discarded, + /// 1 otherwise, as basic_json::size() + size_type size() const noexcept + { + switch (type()) + { + case value_t::null: + case value_t::discarded: + return 0; + case value_t::array: + case value_t::object: + return m_node->len; + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + default: + return 1; + } + } + + /// as basic_json::empty() + bool empty() const noexcept + { + switch (type()) + { + case value_t::null: + case value_t::discarded: + return true; + case value_t::array: + case value_t::object: + return m_node->len == 0; + case value_t::string: + case value_t::boolean: + case value_t::number_integer: + case value_t::number_unsigned: + case value_t::number_float: + case value_t::binary: + default: + return false; + } + } + + ///////////////// + // materialize // + ///////////////// + + /// the basic_json value of this subtree, as parse() would produce it + /// (a discarded value for a discarded view) + BasicJsonType materialize() const + { + if (m_node == nullptr) + { + return BasicJsonType(value_t::discarded); + } + return detail::view::materialize(*m_doc, m_node); + } + + /// byte offset of this value in the source text (for strings: of the + /// first byte after the opening quote); static_cast(-1) for + /// a discarded view and for strings with escapes, which are decoded + std::size_t source_offset() const noexcept + { + return m_node != nullptr && (m_node->flags & detail::view::node_flags::storage) == 0 + ? m_node->off : static_cast(-1); + } + + private: + template friend class basic_json_document; + + basic_json_view(const document_data* d, const node* n) noexcept + : m_doc(d), m_node(n) + {} + + const document_data* m_doc = nullptr; + const node* m_node = nullptr; +}; + +/*! +@brief a parsed JSON text: owns the node index (and, optionally, the text) + +Borrowed parses keep a pointer to the caller's text, which must outlive the +document. Owned parses (parse_copy, rvalue std::string, streams, and inputs +that are not contiguous byte ranges) keep their own copy. +*/ +template +class basic_json_document +{ + using document_data = detail::view::document_data; + + static_assert(sizeof(typename BasicJsonType::number_integer_t) == 8 && sizeof(typename BasicJsonType::number_unsigned_t) == 8, + "json_view supports 64-bit integer types only"); + + public: + using view_type = basic_json_view; + using value_t = detail::value_t; + + /// an empty (discarded) document + basic_json_document() = default; + basic_json_document(basic_json_document&&) noexcept = default; + basic_json_document& operator=(basic_json_document&&) noexcept = default; + basic_json_document(const basic_json_document&) = delete; + basic_json_document& operator=(const basic_json_document&) = delete; + ~basic_json_document() = default; + + ///////////// + // parsing // + ///////////// + + /// parse a JSON text; contiguous byte inputs are borrowed, everything else + /// (and rvalue std::string) is owned + template + NLOHMANN_VIEW_NODISCARD + static basic_json_document parse(InputType&& input, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false) + { + basic_json_document d; + d.read(std::forward(input), allow_exceptions, ignore_comments, ignore_trailing_commas); + return d; + } + + /// parse [first, last) + template::iterator_category>::value, int>::type = 0> + NLOHMANN_VIEW_NODISCARD + static basic_json_document parse(IteratorType first, IteratorType last, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false) + { + basic_json_document d; + d.read_range(first, last, allow_exceptions, ignore_comments, ignore_trailing_commas); + return d; + } + + /// parse a copy of the input; the document does not depend on it afterwards + template + NLOHMANN_VIEW_NODISCARD + static basic_json_document parse_copy(InputType&& input, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false) + { + basic_json_document d; + d.build_owned(collect(std::forward(input)), allow_exceptions, ignore_comments, ignore_trailing_commas); + return d; + } + + /// check whether the input is valid JSON (the result of basic_json::accept) + template + static bool accept(InputType&& input, const bool ignore_comments = false, const bool ignore_trailing_commas = false) + { + basic_json_document d; + d.read(std::forward(input), false, ignore_comments, ignore_trailing_commas); + return !d.is_discarded(); + } + + /// parse into this document, reusing its memory + template + // flawfinder: ignore (a member function, not POSIX read()) + void read(InputType&& input, + const bool allow_exceptions = true, + const bool ignore_comments = false, + const bool ignore_trailing_commas = false) + { + read_kind(std::forward(input), allow_exceptions, ignore_comments, ignore_trailing_commas, + std::integral_constant::value> {}); + } + + //////////// + // access // + //////////// + + /// the root value (discarded if parsing failed without exceptions) + view_type root() const noexcept + { + if (!m_data || m_data->discarded) + { + return view_type(); + } + return view_type(m_data.get(), m_data->tape); + } + + bool is_discarded() const noexcept + { + return !m_data || m_data->discarded; + } + + /// the parsed text + typename view_type::string_view_t source() const noexcept + { + return m_data ? typename view_type::string_view_t(m_data->src, m_data->size) : typename view_type::string_view_t(); + } + + /// whether the document holds its own copy of the text + bool owns_source() const noexcept + { + return m_data && !m_data->owned.empty() && m_data->src == m_data->owned.data(); + } + + /// number of index nodes (values plus object keys) + std::size_t node_count() const noexcept + { + return m_data ? m_data->tape_size : 0; + } + + /// bytes held by the document (index, decoded strings, owned text) + std::size_t memory_usage() const noexcept + { + if (!m_data) + { + return 0; + } + return sizeof(document_data) + (m_data->inline_cap * sizeof(detail::view::node)) + + (m_data->tape != m_data->inline_tape ? m_data->tape_cap * sizeof(detail::view::node) : 0) + + m_data->arena.capacity() + m_data->owned.capacity(); + } + + /// release unused capacity of the index and the decoded strings; like + /// std::vector::shrink_to_fit, this invalidates the views of the document + /// (take new ones from root()) + void shrink_to_fit() + { + if (!m_data) + { + return; + } + using detail::view::node; + document_data& d = *m_data; + + // allocate everything first, so that an exception leaves the document + // unchanged + const bool shrink_arena = d.arena.capacity() > d.arena.size(); + std::string arena(shrink_arena ? d.arena : std::string()); + const bool shrink_tape = d.tape != d.inline_tape && d.tape_size != d.tape_cap; + const bool into_header = d.tape_size <= d.inline_cap; + node* fresh = (shrink_tape && !into_header) ? static_cast(::operator new (d.tape_size * sizeof(node))) : d.inline_tape; + + if (shrink_tape) + { + std::memcpy(fresh, d.tape, d.tape_size * sizeof(node)); + ::operator delete (d.tape); + d.tape = fresh; + d.tape_cap = into_header ? d.inline_cap : d.tape_size; + } + if (shrink_arena) + { + d.arena.swap(arena); + d.base[1] = d.arena.data(); + } + } + + private: + using input_kind = detail::view::input_kind; + + /// create the storage (sized for the input) on first use + void ensure_data(const char* src, std::size_t size) + { + if (!m_data) + { + m_data.reset(document_data::create(detail::view::estimate_nodes(src, size))); + } + } + + /// parse a buffer the document takes ownership of + void build_owned(std::string&& buf, bool allow_exceptions, bool comments, bool trailing_commas) + { + ensure_data(buf.data(), buf.size()); + m_data->owned = std::move(buf); + build(m_data->owned.data(), m_data->owned.size(), allow_exceptions, comments, trailing_commas, true, true); + } + + /// sentinel: src[size] is readable and 0 (std::string, C strings) + void build(const char* src, std::size_t size, bool allow_exceptions, bool comments, bool trailing_commas, bool owned, bool sentinel) + { + ensure_data(src, size); + document_data& d = *m_data; + if (!owned) + { + d.owned.clear(); + } + d.src = src; + d.size = size; + d.tape_size = 0; + d.arena.clear(); + d.discarded = true; + detail::view::parse_failure failure; + bool ok = false; + if (NLOHMANN_VIEW_UNLIKELY(size >= 0xFFFFFFF0u)) + { + failure.code = detail::view::error_code::input_too_large; // LCOV_EXCL_LINE (4 GiB) + } + else + { + ok = detail::view::build < typename BasicJsonType::number_float_t, !detail::abi_config::strict_nul_handling > (d, src, size, comments, trailing_commas, sentinel, failure); + } + if (NLOHMANN_VIEW_LIKELY(ok)) + { + d.base[0] = d.src; + d.base[1] = d.arena.data(); + d.discarded = false; + return; + } + if (allow_exceptions) + { + detail::view::throw_parse_failure(failure, src, size, comments, trailing_commas); + } + } + + // --- input dispatch (see detail::view::input_kind) --- + + void read_kind(std::string&& s, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + build_owned(std::move(s), ae, c, tc); + } + + template + void read_kind(CharT* s, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + static_assert(sizeof(CharT) == 1 && std::is_integral::type>::value, "json_view parses byte (char-like) input"); + if (s == nullptr) + { + build("", 0, ae, c, tc, false, true); + return; + } + const char* cs = reinterpret_cast(s); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + // C strings are null-terminated, as for json::parse(const char*) + // flawfinder: ignore + build(cs, std::strlen(cs), ae, c, tc, false, true); + } + + template + void read_kind(Array& a, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + using CharT = typename std::remove_cv::type>::type; + static_assert(sizeof(CharT) == 1 && std::is_integral::value, "json_view parses byte (char-like) input"); + const std::size_t n = std::extent::value; + // a trailing NUL (string literals) is not part of the text, as for + // parse(), and serves as sentinel + const bool terminated = n > 0 && a[n - 1] == 0 && (!detail::abi_config::strict_nul_handling || std::is_same::value); + build(reinterpret_cast(&a[0]), terminated ? n - 1 : n, ae, c, tc, false, terminated); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + template + void read_kind(const T& s, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + // std::basic_string guarantees data()[size()] == 0: use it as sentinel + const auto size = static_cast(s.size()); + build(size == 0 ? "" : reinterpret_cast(s.data()), size, ae, c, tc, false, // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + detail::view::is_std_string::value || size == 0); + } + + template + void read_kind(const T& s, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + build_owned(std::string(reinterpret_cast(s.data()), static_cast(s.size())), ae, c, tc); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + template + void read_kind(T&& input, bool ae, bool c, bool tc, std::integral_constant /*unused*/) + { + build_owned(detail::view::collect_adapter(detail::input_adapter(std::forward(input))), ae, c, tc); + } + + template + void read_range(IteratorType first, IteratorType last, bool ae, bool c, bool tc) + { + using value_type = typename std::remove_cv::value_type>::type; + // borrow the range where the library's input adapter scans it as one + // block: pointers and, in C++20, contiguous iterators (std::vector, + // std::string, ...) over single bytes + read_range_impl(first, last, ae, c, tc, std::integral_constant < bool, std::is_integral::value + && detail::iterator_input_adapter::supports_bulk_scan > {}); + } + + template + void read_range_impl(IteratorType first, IteratorType last, bool ae, bool c, bool tc, std::true_type /*contiguous bytes*/) + { + const auto size = static_cast(std::distance(first, last)); + build(size == 0 ? "" : reinterpret_cast(&*first), size, ae, c, tc, false, size == 0); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + template + void read_range_impl(IteratorType first, IteratorType last, bool ae, bool c, bool tc, std::false_type /*other*/) + { + build_owned(detail::view::collect_adapter(detail::input_adapter(first, last)), ae, c, tc); + } + + template + static std::string collect(T&& input) + { + return collect_impl(std::forward(input), std::integral_constant::type>::value> {}); + } + + template + static std::string collect_impl(const T& s, std::true_type /*contiguous*/) + { + return {reinterpret_cast(s.data()), static_cast(s.size())}; // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + } + + template + static std::string collect_impl(T&& s, std::false_type /*other*/) + { + return detail::view::collect_adapter(detail::input_adapter(std::forward(s))); + } + + std::unique_ptr m_data{}; // NOLINT(readability-redundant-member-init) +}; + +/// a parsed JSON text for json +using json_document = basic_json_document; +/// a value of a json_document +using json_view = basic_json_view; +/// a parsed JSON text for ordered_json +using ordered_json_document = basic_json_document; +/// a value of an ordered_json_document +using ordered_json_view = basic_json_view; + +NLOHMANN_JSON_NAMESPACE_END + +// #include +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + + + +// undefine the macros of detail/view/macro_scope.hpp (at the end of json_view.hpp) + +#undef NLOHMANN_VIEW_HAS_CPP_17 +#undef NLOHMANN_VIEW_LIKELY +#undef NLOHMANN_VIEW_UNLIKELY +#undef NLOHMANN_VIEW_ALWAYS_INLINE +#undef NLOHMANN_VIEW_NOINLINE +#undef NLOHMANN_VIEW_NODISCARD +#undef NLOHMANN_VIEW_THROW +#undef NLOHMANN_VIEW_LITTLE_ENDIAN +#undef NLOHMANN_VIEW_REPEAT16 + + +#endif // INCLUDE_NLOHMANN_JSON_VIEW_HPP_ diff --git a/src/modules/json.cppm b/src/modules/json.cppm index a32b97603..940c44ab0 100644 --- a/src/modules/json.cppm +++ b/src/modules/json.cppm @@ -19,6 +19,7 @@ module; #include #include +#include export module nlohmann.json; @@ -27,9 +28,15 @@ NLOHMANN_JSON_NAMESPACE_BEGIN using NLOHMANN_JSON_NAMESPACE::adl_serializer; using NLOHMANN_JSON_NAMESPACE::basic_json; +using NLOHMANN_JSON_NAMESPACE::basic_json_document; +using NLOHMANN_JSON_NAMESPACE::basic_json_view; using NLOHMANN_JSON_NAMESPACE::json; +using NLOHMANN_JSON_NAMESPACE::json_document; using NLOHMANN_JSON_NAMESPACE::json_pointer; +using NLOHMANN_JSON_NAMESPACE::json_view; using NLOHMANN_JSON_NAMESPACE::ordered_json; +using NLOHMANN_JSON_NAMESPACE::ordered_json_document; +using NLOHMANN_JSON_NAMESPACE::ordered_json_view; using NLOHMANN_JSON_NAMESPACE::ordered_map; using NLOHMANN_JSON_NAMESPACE::to_string; diff --git a/tests/Makefile b/tests/Makefile index 6bfa14397..e8e61e603 100644 --- a/tests/Makefile +++ b/tests/Makefile @@ -10,12 +10,15 @@ CXXFLAGS += -std=c++11 CPPFLAGS += -I ../single_include FUZZER_ENGINE = src/fuzzer-driver_afl.cpp -FUZZERS = parse_afl_fuzzer parse_bson_fuzzer parse_cbor_fuzzer parse_msgpack_fuzzer parse_ubjson_fuzzer parse_bjdata_fuzzer parse_bon8_fuzzer +FUZZERS = parse_afl_fuzzer parse_bson_fuzzer parse_cbor_fuzzer parse_msgpack_fuzzer parse_ubjson_fuzzer parse_bjdata_fuzzer parse_bon8_fuzzer parse_json_view_fuzzer fuzzers: $(FUZZERS) parse_afl_fuzzer: $(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-parse_json.cpp -o $@ +parse_json_view_fuzzer: + $(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-parse_json_view.cpp -o $@ + parse_bson_fuzzer: $(CXX) $(CXXFLAGS) $(CPPFLAGS) $(FUZZER_ENGINE) src/fuzzer-parse_bson.cpp -o $@ diff --git a/tests/benchmarks/CMakeLists.txt b/tests/benchmarks/CMakeLists.txt index 4d7265145..6f555e20a 100644 --- a/tests/benchmarks/CMakeLists.txt +++ b/tests/benchmarks/CMakeLists.txt @@ -38,3 +38,10 @@ target_compile_features(json_benchmarks PRIVATE cxx_std_11) target_link_libraries(json_benchmarks benchmark ${CMAKE_THREAD_LIBS_INIT}) add_dependencies(json_benchmarks download_test_data) target_include_directories(json_benchmarks PRIVATE ${JSON_BENCHMARK_INCLUDE_DIR} ${CMAKE_BINARY_DIR}/include) + +# the json_document benchmarks, unless the header to benchmark predates json_view.hpp +if(EXISTS "${JSON_BENCHMARK_INCLUDE_DIR}/nlohmann/json_view.hpp") + target_sources(json_benchmarks PRIVATE src/benchmarks_view.cpp) +else() + message(STATUS "${JSON_BENCHMARK_INCLUDE_DIR} has no nlohmann/json_view.hpp: skipping the json_document benchmarks") +endif() diff --git a/tests/benchmarks/README.md b/tests/benchmarks/README.md index aaf5abe46..e329a6c8b 100644 --- a/tests/benchmarks/README.md +++ b/tests/benchmarks/README.md @@ -9,6 +9,7 @@ Micro-benchmarks for parsing, serialization and the binary formats, written with | benchmark | what it does | |---|---| | `ParseFile`, `ParseString` | parse JSON from a file stream or a string | +| `Accept` | validate JSON from a string (`json::accept`) | | `ParseIndented` | parse the large files re-indented by 4 spaces, for the lexer's whitespace handling | | `Dump` | serialize, compact (`-`) and indented (`4`) | | `ToCbor`, `BinaryToCbor` | write CBOR; `BinaryToCbor` writes binary values of growing size | @@ -16,6 +17,10 @@ Micro-benchmarks for parsing, serialization and the binary formats, written with | `FromBinaryBuffer`, `FromBinaryFile` | read CBOR, MessagePack, UBJSON, BJData and BSON from a buffer or a `FILE*` | | `FromBinaryShape` | read deeply nested, container-heavy and scalar-heavy documents in every binary format | | `FromCborChunkedString` | read CBOR strings split into indefinite-length chunks | +| `ViewParse`, `ViewRead` | parse with [`json_document`](https://json.nlohmann.me/features/json_view/) into a new document, or into one that is reused; compare with `ParseString` | +| `ViewParseIndented` | as `ParseIndented`, with a reused `json_document` | +| `ViewAccept` | validate with `json_document::accept`; compare with `Accept` | +| `ViewMaterialize` | convert a parsed `json_document` into a `json` value | The input files are those of [nativejson-benchmark](https://github.com/miloyip/nativejson-benchmark) (`canada`, `citm_catalog`, `twitter`), a large `jeopardy` file, and number-heavy files (`floats`, `signed_ints`, ...). @@ -110,7 +115,9 @@ Python, unpinned versions work as well. In its output: - `OVERALL_GEOMEAN` summarizes all benchmarks; - `-a` shows only the aggregates, not every repetition. -The header you compare with must support everything the benchmarks use. The current benchmarks build against 3.12.0. +The header you compare with must support everything the benchmarks use. The current benchmarks build against 3.12.0; +the `View*` benchmarks (in `src/benchmarks_view.cpp`) are only built if the directory also holds +`nlohmann/json_view.hpp`. Only benchmarks present in both result files are compared, so for older releases, either filter the benchmarks or build that release's own `tests/benchmarks` against its own header. diff --git a/tests/benchmarks/src/benchmarks.cpp b/tests/benchmarks/src/benchmarks.cpp index 5df7dd473..68635ff09 100644 --- a/tests/benchmarks/src/benchmarks.cpp +++ b/tests/benchmarks/src/benchmarks.cpp @@ -81,6 +81,31 @@ BENCHMARK_CAPTURE(ParseString, signed_ints, TEST_DATA_DIRECTORY "/regressi BENCHMARK_CAPTURE(ParseString, unsigned_ints, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json"); BENCHMARK_CAPTURE(ParseString, small_signed_ints, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json"); +////////////////////////////////////////////////////////////////////////////// +// validate JSON from string +////////////////////////////////////////////////////////////////////////////// + +static void Accept(benchmark::State& state, const char* filename) +{ + std::ifstream f(filename); + std::string str((std::istreambuf_iterator(f)), std::istreambuf_iterator()); + + while (state.KeepRunning()) + { + benchmark::DoNotOptimize(json::accept(str)); + } + + state.SetBytesProcessed(state.iterations() * str.size()); +} +BENCHMARK_CAPTURE(Accept, jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json"); +BENCHMARK_CAPTURE(Accept, canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json"); +BENCHMARK_CAPTURE(Accept, citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json"); +BENCHMARK_CAPTURE(Accept, twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json"); +BENCHMARK_CAPTURE(Accept, floats, TEST_DATA_DIRECTORY "/regression/floats.json"); +BENCHMARK_CAPTURE(Accept, signed_ints, TEST_DATA_DIRECTORY "/regression/signed_ints.json"); +BENCHMARK_CAPTURE(Accept, unsigned_ints, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json"); +BENCHMARK_CAPTURE(Accept, small_signed_ints, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json"); + ////////////////////////////////////////////////////////////////////////////// // parse pretty-printed JSON from string // diff --git a/tests/benchmarks/src/benchmarks_view.cpp b/tests/benchmarks/src/benchmarks_view.cpp new file mode 100644 index 000000000..d6d04d1d5 --- /dev/null +++ b/tests/benchmarks/src/benchmarks_view.cpp @@ -0,0 +1,146 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Benchmarks of json_document (nlohmann/json_view.hpp). They use the inputs of +// ParseString and ParseIndented in benchmarks.cpp, so that each ViewParse row +// can be read against the json::parse row of the same file. + +#include +#include +#include +#include +#include +#include + +using json = nlohmann::json; +using json_document = nlohmann::json_document; + +static std::string read_file(const char* filename) +{ + std::ifstream f(filename, std::ios::binary); + return std::string((std::istreambuf_iterator(f)), std::istreambuf_iterator()); +} + +#define JSON_VIEW_BENCHMARK_FILES(fn) \ + BENCHMARK_CAPTURE(fn, jeopardy, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json"); \ + BENCHMARK_CAPTURE(fn, canada, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json"); \ + BENCHMARK_CAPTURE(fn, citm_catalog, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json"); \ + BENCHMARK_CAPTURE(fn, twitter, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json"); \ + BENCHMARK_CAPTURE(fn, floats, TEST_DATA_DIRECTORY "/regression/floats.json"); \ + BENCHMARK_CAPTURE(fn, signed_ints, TEST_DATA_DIRECTORY "/regression/signed_ints.json"); \ + BENCHMARK_CAPTURE(fn, unsigned_ints, TEST_DATA_DIRECTORY "/regression/unsigned_ints.json"); \ + BENCHMARK_CAPTURE(fn, small_signed_ints, TEST_DATA_DIRECTORY "/regression/small_signed_ints.json") + +////////////////////////////////////////////////////////////////////////////// +// parse into a new document (compare with ParseString) +////////////////////////////////////////////////////////////////////////////// + +static void ViewParse(benchmark::State& state, const char* filename) +{ + const std::string str = read_file(filename); + + while (state.KeepRunning()) + { + state.PauseTiming(); + auto* d = new json_document(); + state.ResumeTiming(); + + *d = json_document::parse(str); + + state.PauseTiming(); + delete d; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * str.size()); +} +JSON_VIEW_BENCHMARK_FILES(ViewParse); + +////////////////////////////////////////////////////////////////////////////// +// parse into a document that is reused (its memory stays allocated) +////////////////////////////////////////////////////////////////////////////// + +static void ViewRead(benchmark::State& state, const char* filename) +{ + const std::string str = read_file(filename); + json_document d; + + while (state.KeepRunning()) + { + d.read(str); + benchmark::DoNotOptimize(d); + } + + state.SetBytesProcessed(state.iterations() * str.size()); +} +JSON_VIEW_BENCHMARK_FILES(ViewRead); + +////////////////////////////////////////////////////////////////////////////// +// parse pretty-printed JSON (compare with ParseIndented) +////////////////////////////////////////////////////////////////////////////// + +static void ViewParseIndented(benchmark::State& state, const char* filename, int indent) +{ + const std::string indented = json::parse(read_file(filename)).dump(indent); + json_document d; + + while (state.KeepRunning()) + { + d.read(indented); + benchmark::DoNotOptimize(d); + } + + state.SetBytesProcessed(state.iterations() * indented.size()); +} +BENCHMARK_CAPTURE(ViewParseIndented, jeopardy / 4, TEST_DATA_DIRECTORY "/jeopardy/jeopardy.json", 4); +BENCHMARK_CAPTURE(ViewParseIndented, canada / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/canada.json", 4); +BENCHMARK_CAPTURE(ViewParseIndented, citm_catalog / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/citm_catalog.json", 4); +BENCHMARK_CAPTURE(ViewParseIndented, twitter / 4, TEST_DATA_DIRECTORY "/nativejson-benchmark/twitter.json", 4); + +////////////////////////////////////////////////////////////////////////////// +// validate only (compare with Accept) +////////////////////////////////////////////////////////////////////////////// + +static void ViewAccept(benchmark::State& state, const char* filename) +{ + const std::string str = read_file(filename); + + while (state.KeepRunning()) + { + benchmark::DoNotOptimize(json_document::accept(str)); + } + + state.SetBytesProcessed(state.iterations() * str.size()); +} +JSON_VIEW_BENCHMARK_FILES(ViewAccept); + +////////////////////////////////////////////////////////////////////////////// +// convert a parsed document into a json value +////////////////////////////////////////////////////////////////////////////// + +static void ViewMaterialize(benchmark::State& state, const char* filename) +{ + const std::string str = read_file(filename); + const json_document d = json_document::parse(str); + + while (state.KeepRunning()) + { + state.PauseTiming(); + auto* j = new json(); + state.ResumeTiming(); + + *j = d.root().materialize(); + + state.PauseTiming(); + delete j; + state.ResumeTiming(); + } + + state.SetBytesProcessed(state.iterations() * str.size()); +} +JSON_VIEW_BENCHMARK_FILES(ViewMaterialize); diff --git a/tests/fuzzing.md b/tests/fuzzing.md index b7c2da27f..59cdb9aca 100644 --- a/tests/fuzzing.md +++ b/tests/fuzzing.md @@ -3,6 +3,12 @@ Each parser of the library (JSON, BJData, BON8, BSON, CBOR, MessagePack, and UBJSON) can be fuzz tested. Currently, [libFuzzer](https://llvm.org/docs/LibFuzzer.html) and [afl++](https://github.com/AFLplusplus/AFLplusplus) are supported. +Additionally, `parse_json_view_fuzzer` (`tests/src/fuzzer-parse_json_view.cpp`) cross-checks `json_document`/`json_view` +(the zero-copy, read-only view declared in `json_view.hpp`) against `basic_json` on the same JSON text: it asserts that +`json_document::accept` agrees with `json::accept`, that an accepted input materializes to the same value `json::parse` +produces, and that a rejected input makes both parsers throw with an identical `what()`. It takes plain JSON text, so it +reuses the `corpus_json` corpus rather than a format of its own. + ## Corpus creation For most effective fuzzing, a [corpus](https://llvm.org/docs/LibFuzzer.html#corpus) should be provided. A corpus is a diff --git a/tests/module_cpp20/main.cpp b/tests/module_cpp20/main.cpp index a844e657d..4e492cef9 100644 --- a/tests/module_cpp20/main.cpp +++ b/tests/module_cpp20/main.cpp @@ -52,8 +52,13 @@ int main() std::ostringstream os; os << j << oj << lit; + // json_document / json_view (json_view.hpp): zero-copy parse of a literal + const nlohmann::json_document doc = nlohmann::json_document::parse(R"({"a": 1, "list": [1, 2, 3]})"); + const std::size_t doc_size = doc.root().size(); + // use every result so the references cannot be optimized away return (a == 1 && last == 3 && b == 2 && lit.size() == 3 - && m.size() == 1 && !dumped.empty() && !os.str().empty()) + && m.size() == 1 && !dumped.empty() && !os.str().empty() + && doc_size == 2) ? 0 : 1; } diff --git a/tests/src/fuzzer-parse_json_view.cpp b/tests/src/fuzzer-parse_json_view.cpp new file mode 100644 index 000000000..59e32a556 --- /dev/null +++ b/tests/src/fuzzer-parse_json_view.cpp @@ -0,0 +1,100 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +/* +This file implements a parser test suitable for fuzz testing. It checks that +json_document (the zero-copy, read-only view of a parsed JSON text declared in +json_view.hpp) agrees with basic_json on every input: + +- json_document::accept(data) must equal json::accept(data) +- if the input is accepted, json_document::parse(data).root().materialize() + must equal json::parse(data) +- if the input is rejected, json_document::parse(data) (with exceptions + enabled) must throw a json::parse_error or json::out_of_range whose what() + is identical to the one json::parse(data) throws + +The provided function `LLVMFuzzerTestOneInput` can be used in different fuzzer +drivers. +*/ + +#include +#include +#include +#include + +// the checks below are assertions; NDEBUG would compile them away +#ifdef NDEBUG + #error "the fuzzer drivers must be built without NDEBUG" +#endif + +using json = nlohmann::json; +using json_document = nlohmann::json_document; + +// see http://llvm.org/docs/LibFuzzer.html +extern "C" int LLVMFuzzerTestOneInput(const uint8_t* data, size_t size) +{ + // json_document::accept only has a single-argument overload; wrap the raw + // bytes in a (borrowed) std::string so the same bytes can be handed to it + const std::string input(reinterpret_cast(data), size); // NOLINT(cppcoreguidelines-pro-type-reinterpret-cast) + + const bool accepted_by_json = json::accept(data, data + size); + const bool accepted_by_view = json_document::accept(input); + + // json_document::accept must agree with json::accept on every input + assert(accepted_by_json == accepted_by_view); + + if (accepted_by_json) + { + // both parsers must agree on the resulting value + json const j1 = json::parse(data, data + size); + json_document const doc = json_document::parse(input); + json const j2 = doc.root().materialize(); + assert(j1 == j2); + } + else + { + // both parsers must reject the input the same way when exceptions are used + std::string expected_what; + bool json_threw = false; + try + { + static_cast(json::parse(data, data + size)); + } + catch (const json::parse_error& e) + { + expected_what = e.what(); + json_threw = true; + } + catch (const json::out_of_range& e) + { + expected_what = e.what(); + json_threw = true; + } + assert(json_threw); + + bool view_threw = false; + try + { + static_cast(json_document::parse(input)); + } + catch (const json::parse_error& e) + { + assert(e.what() == expected_what); + view_threw = true; + } + catch (const json::out_of_range& e) + { + assert(e.what() == expected_what); + view_threw = true; + } + assert(view_threw); + } + + // return 0 - non-zero return values are reserved for future use + return 0; +} diff --git a/tests/src/unit-json_view.cpp b/tests/src/unit-json_view.cpp new file mode 100644 index 000000000..d953d9903 --- /dev/null +++ b/tests/src/unit-json_view.cpp @@ -0,0 +1,413 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +using nlohmann::json; +using nlohmann::ordered_json; +using nlohmann::json_document; +using nlohmann::json_view; +using nlohmann::ordered_json_document; + +#include +#include +#include +#include +#include +#include +#include +#include + +#ifdef JSON_HAS_CPP_17 + #include +#endif + +namespace +{ +#if !defined(JSON_NOEXCEPTION) +// the exception parse() throws for a text, or "" if it accepts it +std::string parse_exception(const std::string& text, bool comments = false, bool trailing_commas = false) +{ + try + { + const json j = json::parse(text, nullptr, true, comments, trailing_commas); + static_cast(j); + } + catch (const json::exception& e) + { + return e.what(); + } + return ""; +} + +std::string view_exception(const std::string& text, bool comments = false, bool trailing_commas = false) +{ + try + { + const json_document d = json_document::parse(text, true, comments, trailing_commas); + static_cast(d); + } + catch (const json::exception& e) + { + return e.what(); + } + return ""; +} +#endif + +// a small deterministic generator of documents +struct generator +{ + std::mt19937 rng{5295}; // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + + int r(int n) + { + return static_cast(rng() % static_cast(n)); + } + + void str(std::string& o) + { + static const char* const pieces[] = {"a", "Z", " ", "\\n", "\\\"", "\\u00e9", "\\ud83d\\ude00", "\xc3\xa9", "\xe3\x81\x82", "long text beyond the first sixteen bytes"}; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + o += '"'; + for (int n = r(5); n > 0; --n) + { + o += pieces[r(10)]; + } + o += '"'; + } + + void value(std::string& o, int depth) + { + static const char* const scalars[] = {"0", "-1", "123456789012", "18446744073709551615", "18446744073709551616", "-9223372036854775809", // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + "1.5", "-2.25e-3", "1E2", "0.1", "true", "false", "null" + }; + const int k = depth > 5 ? 2 + r(4) : r(6); + if (k < 2) + { + const bool object = k == 0; + o += object ? '{' : '['; + for (int i = r(5); i > 0; --i) + { + if (object) + { + str(o); + o += r(2) == 0 ? ":" : " : "; + } + value(o, depth + 1); + o += i > 1 ? ", " : ""; + } + o += object ? '}' : ']'; + } + else if (k < 4) + { + str(o); + } + else + { + o += scalars[r(13)]; + } + } +}; +} // namespace + +TEST_CASE("json_view") +{ + SECTION("types and capacity") + { + for (const char* text : + {"null", "true", "false", "0", "-1", "18446744073709551615", "-9223372036854775808", "18446744073709551616", "1.5", + "\"\"", "\"text\"", "[]", "[1,2,3]", "{}", "{\"a\":1,\"b\":2}" // NOLINT(modernize-raw-string-literal) + }) + { + CAPTURE(text) + const json j = json::parse(text); + const json_document d = json_document::parse(text); + const json_view v = d.root(); + CHECK(v.type() == j.type()); + CHECK(v.is_null() == j.is_null()); + CHECK(v.is_boolean() == j.is_boolean()); + CHECK(v.is_number() == j.is_number()); + CHECK(v.is_number_integer() == j.is_number_integer()); + CHECK(v.is_number_unsigned() == j.is_number_unsigned()); + CHECK(v.is_number_float() == j.is_number_float()); + CHECK(v.is_string() == j.is_string()); + CHECK(v.is_array() == j.is_array()); + CHECK(v.is_object() == j.is_object()); + CHECK(v.is_binary() == j.is_binary()); + CHECK(v.is_primitive() == j.is_primitive()); + CHECK(v.is_structured() == j.is_structured()); + CHECK(!v.is_discarded()); + CHECK(static_cast(v)); + CHECK(v.size() == j.size()); + CHECK(v.empty() == j.empty()); + CHECK(v.materialize() == j); + } + + const json_view invalid{}; + CHECK(invalid.is_discarded()); + CHECK(!static_cast(invalid)); + CHECK(invalid.type() == json::value_t::discarded); + CHECK(invalid.size() == 0); + CHECK(invalid.empty()); + CHECK(invalid.materialize().is_discarded()); + CHECK(invalid.source_offset() == static_cast(-1)); + } + + SECTION("materialize") + { + generator g; + for (int i = 0; i < 2000; ++i) + { + std::string text; + g.value(text, 0); + CAPTURE(text) + CHECK(json_document::parse(text).root().materialize() == json::parse(text)); + // member order as ordered_json::parse keeps it + CHECK(ordered_json_document::parse(text).root().materialize().dump() == ordered_json::parse(text).dump()); + } + // duplicate keys: the last value, at the position of the first key + CHECK(json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize() == json::parse(R"({"a":1,"b":2,"a":3})")); + CHECK(ordered_json_document::parse(R"({"a":1,"b":2,"a":3})").root().materialize().dump() == R"({"a":3,"b":2})"); + // very deep nesting (iterative, as parse()) + const std::string deep = std::string(100000, '[') + std::string(100000, ']'); + CHECK(json_document::parse(deep).root().materialize() == json::parse(deep)); +#if JSON_DIAGNOSTICS + // the parents are set, so errors name the path + const json m = json_document::parse(R"({"a":{"b":[1]}})").root().materialize(); + CHECK_THROWS_WITH_AS(m.at("a").at("b").at(0).at("x"), "[json.exception.type_error.304] (/a/b/0) cannot use at() with number", json::type_error&); +#endif + } + + SECTION("parse errors are those of parse()") + { + for (const char* text : + { + "", " ", "[", "]", "{", "[1,]", "{\"a\":1,}", "[1 2]", "{\"a\" 1}", "{1:2}", "tru", "nul", "fals", "truex", "-", "01", "1.", ".5", "1e", + "\"", "\"abc", "\"\\x\"", "\"\\u12\"", "\"\\ud800\"", "\"\\udc00\"", "\"\x01\"", "\"\xff\"", "\"\xc3\"", "[1]x", "/", "/*", "[\n 1,\n x\n]", // NOLINT(modernize-raw-string-literal) + "1e400", "-1e400", "[1.7976931348623159e308]", "{\"a\":\n{\"b\": [1, 2,\n 3 x]}}" + }) + { + CAPTURE(text) +#if !defined(JSON_NOEXCEPTION) + const std::string expected = parse_exception(text); + REQUIRE(!expected.empty()); + CHECK(view_exception(text) == expected); +#endif + CHECK(!json_document::accept(text)); + const json_document d = json_document::parse(text, false); + CHECK(d.is_discarded()); + CHECK(d.root().is_discarded()); + CHECK(d.node_count() == 0); + } + // the exception types + json_document _; + CHECK_THROWS_AS(_ = json_document::parse("[1,"), json::parse_error&); + CHECK_THROWS_AS(_ = json_document::parse("1e400"), json::out_of_range&); + } + + SECTION("parse options") + { + for (const char* text : + {"// c\n[1]", "[1, /* c */ 2]", "[1,]", "{\"a\":1,}", "[1,/* c */]", "/", "/* ", "[1,,]" + }) + { + CAPTURE(text) + for (int options = 0; options < 4; ++options) + { + const bool comments = (options & 1) != 0; + const bool trailing_commas = (options & 2) != 0; + CHECK(json_document::accept(text, comments, trailing_commas) == json::accept(text, comments, trailing_commas)); +#if !defined(JSON_NOEXCEPTION) + CHECK(view_exception(text, comments, trailing_commas) == parse_exception(text, comments, trailing_commas)); +#endif + } + } + } + + SECTION("overflow of the floating-point type") + { + // the overflow is that of the document's number_float_t: 1e39 + // overflows a float, the largest float 3.4028235e38 does not, but + // 3.4028236e38 rounds beyond it; a double document is not affected + using json_float = nlohmann::basic_json; + using float_document = nlohmann::basic_json_document; + CHECK_FALSE(json_float::accept("1e39")); + CHECK(json_float::accept("3.4028235e38")); + CHECK_FALSE(json_float::accept("3.4028236e38")); + for (const char* text : + { + "1e39", "-1e39", "[3.4028235e38]", "[3.4028236e38]", "{\"a\": [1e38, -3.4028235e38]}", "{\"a\": [1e-50, 1e39]}" + }) + { + CAPTURE(text) + const bool accepted = json_float::accept(text); + CHECK(float_document::accept(text) == accepted); + CHECK(float_document::parse(text, false).is_discarded() == !accepted); + CHECK(json_document::accept(text)); + if (accepted) + { + CHECK(float_document::parse(text).root().materialize() == json_float::parse(text)); + } + } + float_document f; + CHECK_THROWS_WITH_AS(f = float_document::parse("1e39"), "[json.exception.out_of_range.406] number overflow parsing '1e39'", json::out_of_range&); + CHECK_THROWS_WITH_AS(f = float_document::parse("[3.4028236e38]"), "[json.exception.out_of_range.406] number overflow parsing '3.4028236e38'", json::out_of_range&); + } + + SECTION("NUL and BOM") + { + const std::string with_nul("[1]\0garbage", 11); + CHECK(json_document::accept(with_nul) == json::accept(with_nul)); + const std::string nul_in_comment("[1, // c\0\n2]", 12); + CHECK(json_document::accept(nul_in_comment, true) == json::accept(nul_in_comment, true)); + CHECK(json_document::parse("\xEF\xBB\xBF[1]").root().materialize() == json::parse("\xEF\xBB\xBF[1]")); +#if !defined(JSON_NOEXCEPTION) + CHECK(view_exception("\xEF\xBB") == parse_exception("\xEF\xBB")); +#endif + } + + SECTION("inputs") + { + const std::string text = R"([1, "two", {"three": 3.5}])"; + const json expected = json::parse(text); + + // borrowed: the text must outlive the document + const json_document borrowed = json_document::parse(text); + CHECK(!borrowed.owns_source()); + CHECK(borrowed.source().data() == text.data()); + CHECK(borrowed.root().materialize() == expected); + CHECK(json_document::parse(text.c_str()).root().materialize() == expected); + CHECK(json_document::parse(R"([1, "two", {"three": 3.5}])").root().materialize() == expected); + CHECK(json_document::parse(text.data(), text.data() + text.size()).root().materialize() == expected); + const std::vector chars(text.begin(), text.end()); + CHECK(!json_document::parse(chars).owns_source()); + CHECK(json_document::parse(chars).root().materialize() == expected); + const std::vector bytes(text.begin(), text.end()); + CHECK(json_document::parse(bytes).root().materialize() == expected); +#ifdef JSON_HAS_CPP_17 + const std::string_view sv = text; + CHECK(!json_document::parse(sv).owns_source()); + CHECK(json_document::parse(sv).root().materialize() == expected); +#endif + + // owned + std::string moved = text; + const json_document from_rvalue = json_document::parse(std::move(moved)); + CHECK(from_rvalue.owns_source()); + CHECK(from_rvalue.root().materialize() == expected); + CHECK(json_document::parse(std::vector(text.begin(), text.end())).owns_source()); + CHECK(json_document::parse_copy(text).owns_source()); + CHECK(json_document::parse_copy(text).root().materialize() == expected); + std::istringstream stream(text); + const json_document from_stream = json_document::parse(stream); + CHECK(from_stream.owns_source()); + CHECK(from_stream.root().materialize() == expected); + const std::list list(text.begin(), text.end()); + CHECK(json_document::parse(list.begin(), list.end()).owns_source()); + CHECK(json_document::parse(list.begin(), list.end()).root().materialize() == expected); + + // iterator pairs: pointers are borrowed, and so are contiguous library + // iterators where the input adapter detects them (C++20) + CHECK(!json_document::parse(text.data(), text.data() + text.size()).owns_source()); + const bool contiguous = nlohmann::detail::iterator_input_adapter::const_iterator>::supports_bulk_scan; + const json_document from_iterators = json_document::parse(chars.cbegin(), chars.cend()); + CHECK(from_iterators.owns_source() != contiguous); + CHECK((from_iterators.source().data() == chars.data()) == contiguous); + CHECK(from_iterators.root().materialize() == expected); + const std::string padded = "x" + text + "x"; + CHECK(json_document::parse(padded.begin() + 1, padded.end() - 1).root().materialize() == expected); + CHECK(json_document::parse(chars.cbegin(), chars.cbegin(), false).is_discarded()); + const std::wstring wide = L"[\"\u00e4\u20ac\", 1]"; + CHECK(json_document::parse(wide).root().materialize() == json::parse(wide)); + CHECK(json_document::parse(static_cast(nullptr), false).is_discarded()); + CHECK(json_document::parse("", false).is_discarded()); + } + + SECTION("document lifetime and reuse") + { + json_document d; + CHECK(d.is_discarded()); + CHECK(d.root().is_discarded()); + CHECK(d.node_count() == 0); + CHECK(d.memory_usage() == 0); + CHECK(d.source().empty()); + + const std::string a = "[1,2,3]"; + const std::string b = "{\"x\":[true]}"; + d.read(a); + CHECK(d.node_count() == 4); + CHECK(d.root().materialize() == json::parse(a)); + d.read(b); + CHECK(d.node_count() == 4); + CHECK(d.root().materialize() == json::parse(b)); + d.read("[", false); + CHECK(d.is_discarded()); + + // views stay valid when the document moves + json_document first = json_document::parse(a); + const json_view root = first.root(); + const json_document second = std::move(first); + CHECK(root.materialize() == json::parse(a)); + CHECK(second.root().materialize() == json::parse(a)); + } + + SECTION("memory") + { + std::string big = "["; + for (int i = 0; i < 10000; ++i) + { + big += (i != 0 ? ",\"" : "\"") + std::to_string(i) + "\""; + } + big += ']'; + json_document d = json_document::parse(big); + CHECK(d.node_count() == 10001); + const std::size_t before = d.memory_usage(); + d.shrink_to_fit(); // (invalidates views, like std::vector::shrink_to_fit) + CHECK(d.memory_usage() <= before); + CHECK(d.node_count() == 10001); + CHECK(d.root().materialize() == json::parse(big)); + CHECK(d.root().size() == 10000); + d.shrink_to_fit(); // nothing left to release + + // after reading a smaller text, both the index and the decoded strings + // shrink, and the strings are found in their new place + std::string escaped = "["; + for (int i = 0; i < 1000; ++i) + { + escaped += (i != 0 ? ",\"a\\n" : "\"a\\n") + std::to_string(i) + "\""; + } + escaped += ']'; + json_document reused = json_document::parse(escaped); + const std::string smaller = "[\"x\\ty\", [true, \"\\u00e4\"]]"; // NOLINT(modernize-raw-string-literal) + reused.read(smaller); + const std::size_t grown = reused.memory_usage(); + reused.shrink_to_fit(); + CHECK(reused.memory_usage() < grown); + CHECK(reused.root().materialize() == json::parse(smaller)); + + // an empty document has nothing to release + json_document empty; + empty.shrink_to_fit(); + CHECK(empty.memory_usage() == 0); + + // a small document stays in the storage block of the header + json_document small = json_document::parse("[1,[2,3],{\"a\":\"b\\n\"}]"); // NOLINT(modernize-raw-string-literal) + small.shrink_to_fit(); + CHECK(small.root().materialize() == json::parse("[1,[2,3],{\"a\":\"b\\n\"}]")); + } + + SECTION("source offsets") + { + const std::string text = R"( {"key": "value", "escaped": "a\nb", "n": 42})"; + const json_document d = json_document::parse(text); + CHECK(d.root().source_offset() == 2); + // (element access comes with a later change; the offsets of the + // string nodes are checked through materialize() above) + } +} diff --git a/tests/src/unit-json_view_builder.cpp b/tests/src/unit-json_view_builder.cpp new file mode 100644 index 000000000..9d3ea943b --- /dev/null +++ b/tests/src/unit-json_view_builder.cpp @@ -0,0 +1,403 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +#include +#if JSON_TEST_USING_MULTIPLE_HEADERS + #include + #include +#else + #include // the single header contains the internal headers +#endif +using nlohmann::json; + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace +{ +using nlohmann::detail::view::document_data; +using nlohmann::detail::view::node; + +// the node index of a text; a vector input has no terminating NUL, so that +// AddressSanitizer catches any read past the last byte +struct built +{ + std::unique_ptr data{}; // NOLINT(readability-redundant-member-init) + std::vector copy{}; // NOLINT(readability-redundant-member-init) + bool ok = false; + nlohmann::detail::view::parse_failure failure{}; +}; + +template +built build(const std::string& text, bool comments, bool trailing_commas, bool sentinel) +{ + built r; + r.data.reset(document_data::create(nlohmann::detail::view::estimate_nodes(text.data(), text.size()))); + const char* src = text.c_str(); + if (!sentinel) + { + r.copy.assign(text.begin(), text.end()); + src = r.copy.data(); + } + r.ok = nlohmann::detail::view::build < FloatType, !nlohmann::detail::abi_config::strict_nul_handling > (*r.data, src, text.size(), comments, trailing_commas, sentinel, r.failure); + r.data->src = src; + r.data->base[0] = src; + r.data->base[1] = r.data->arena.data(); + return r; +} + +// the value of a subtree, as json::parse would build it +json value_of(const document_data& d, const node*& n) +{ + const node& x = *n; + ++n; + switch (static_cast(x.kind)) + { + case json::value_t::object: + { + json o = json::object(); + const node* const end = &x + x.next; + while (n != end) + { + const std::string key(d.str(*n), n->len); + ++n; + o[key] = value_of(d, n); + } + return o; + } + case json::value_t::array: + { + json a = json::array(); + const node* const end = &x + x.next; + while (n != end) + { + a.push_back(value_of(d, n)); + } + return a; + } + case json::value_t::string: + return std::string(d.str(x), x.len); + case json::value_t::boolean: + return (x.flags & nlohmann::detail::view::node_flags::is_true) != 0; + case json::value_t::number_integer: + return static_cast(nlohmann::detail::view::integer_bits(x)); + case json::value_t::number_unsigned: + return nlohmann::detail::view::integer_bits(x); + case json::value_t::number_float: + return json::parse(std::string(d.src + x.off, x.len)).get(); + case json::value_t::null: + case json::value_t::binary: + case json::value_t::discarded: + default: + return nullptr; + } +} + +json value_of(const built& b) +{ + const node* n = b.data->tape; + json v = value_of(*b.data, n); + CHECK(n == b.data->tape + b.data->tape_size); + return v; +} + +// accept/reject and the value must match json::parse, for all options and +// with and without a NUL after the text +void check_same(const std::string& text) +{ + CAPTURE(text) + for (int options = 0; options < 4; ++options) + { + const bool comments = (options & 1) != 0; + const bool trailing_commas = (options & 2) != 0; + const bool accepted = json::accept(text, comments, trailing_commas); + for (const bool sentinel : + { + true, false + }) + { + const built b = build(text, comments, trailing_commas, sentinel); + CHECK(b.ok == accepted); + if (b.ok && accepted) + { + CHECK(value_of(b) == json::parse(text, nullptr, true, comments, trailing_commas)); + } + } + } +} + +// a small deterministic generator of documents +struct generator +{ + std::mt19937 rng{5295}; // NOLINT(cert-msc32-c,cert-msc51-cpp,bugprone-random-generator-seed) + + int r(int n) + { + return static_cast(rng() % static_cast(n)); + } + + void ws(std::string& o) + { + for (int n = r(4) == 0 ? r(12) : r(2); n > 0; --n) + { + o += " \n\t\r "[r(6)]; + } + } + + void str(std::string& o) + { + static const char* const pieces[] = {"a", "Z", " ", "~", "\\n", "\\\"", "\\\\", "\\/", "\\u00e9", "\\ud83d\\ude00", "\xc3\xa9", "\xe3\x81\x82", "\xf0\x9f\x98\x80", "\x7f", "\\u001f", "long enough text to leave the first 16 bytes"}; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + o += '"'; + for (int n = r(3) == 0 ? r(20) : r(6); n > 0; --n) + { + o += pieces[r(16)]; + } + o += '"'; + } + + void num(std::string& o) + { + static const char* const numbers[] = {"0", "-0", "1", "-1", "12", "123456789", "1234567890123456789", "9223372036854775807", "-9223372036854775808", // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + "9223372036854775808", "18446744073709551615", "18446744073709551616", "-9223372036854775809", + "1.5", "-2.25e-3", "1e10", "1E+2", "0.000001", "3.141592653589793238462643", "1e308", "-1e-400", "123.456e7" + }; + o += numbers[r(22)]; + } + + void value(std::string& o, int depth) + { + ws(o); + const int k = depth > 5 ? 2 + r(6) : r(8); + if (k == 0 || k == 1) + { + const bool object = k == 0; + o += object ? '{' : '['; + for (int i = r(5); i > 0; --i) + { + ws(o); + if (object) + { + str(o); + ws(o); + o += ':'; + } + value(o, depth + 1); + o += i > 1 ? "," : ""; + } + ws(o); + o += object ? '}' : ']'; + } + else if (k < 4) + { + str(o); + } + else if (k < 6) + { + num(o); + } + else + { + static const char* const literals[] = {"true", "false", "null"}; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + o += literals[r(3)]; + } + ws(o); + } +}; +} // namespace + +TEST_CASE("json_view string_ref") +{ + // std::string_view in C++17, a stand-in with the same members before + using nlohmann::detail::view::string_ref; + const std::string text = "abc"; + const string_ref r(text); + CHECK(r.length() == 3); + CHECK(std::string(r.begin(), r.end()) == "abc"); + CHECK(r[1] == 'b'); + CHECK(r != string_ref("abd")); + CHECK_FALSE(r != string_ref("abcd", 3)); + std::ostringstream o; + o << r; + CHECK(o.str() == "abc"); +} + +TEST_CASE("json_view builder") +{ + SECTION("scalars and containers") + { + for (const char* text : + { + "null", "true", "false", "0", "-0", "42", "-42", "1.5", "\"\"", "\"abc\"", "[]", "{}", "[1,2,3]", "{\"a\":1,\"b\":[true,null]}", // NOLINT(modernize-raw-string-literal) + " [ 1 , 2 ] ", "{\"a\" : {\"b\" : {}}}", "[[[]]]", "\"\\u00e4\\n\\ud83d\\ude00\"", "{\"a\":1,\"a\":2}", "18446744073709551616", // NOLINT(modernize-raw-string-literal) + "-9223372036854775809", "123456789012345678901234567890", "1e400", "-1e400", "1.7976931348623157e308" + }) + { + check_same(text); + } + + // the midpoint between the largest double and 2^1024 rounds to + // infinity (an overflow), one less to the largest double: with more + // than 19 digits, Eisel-Lemire cannot decide these, and the overflow + // check needs the exact comparison with the midpoint + const std::string midpoint = "179769313486231580793728971405303415079934132710037826936173778980444968292764750946649017977587207096330286416692887910946555547851940402630657488671505820681908902000708383676273854845817711531764475730270069855571366959622842914819860834936475292719074168444365510704342711559699508093042880177904174497792"; + const std::string below = "179769313486231580793728971405303415079934132710037826936173778980444968292764750946649017977587207096330286416692887910946555547851940402630657488671505820681908902000708383676273854845817711531764475730270069855571366959622842914819860834936475292719074168444365510704342711559699508093042880177904174497791"; + check_same(midpoint); + check_same("-" + midpoint); + check_same(below); + check_same("[" + below + "," + midpoint + "]"); + + // the check uses the floating-point type of the document: with float, + // the view rejects what parse() rejects (out_of_range.406), and a + // double document is not affected + using float_json = nlohmann::basic_json; + CHECK_FALSE(float_json::accept("1e39")); + CHECK(float_json::accept("3.4028235e38")); + CHECK_FALSE(float_json::accept("3.4028236e38")); + for (const char* text : + { + "1e39", "-1e39", "3.4028235e38", "-3.4028235e38", "3.4028236e38", "-3.4028236e38", "3.4028234663852886e38", "1e38", + "340282356779733661637539395458142568448", "340282356779733661637539395458142568447.99", "0.00034028236e42", + "[1.5e38, 3.5e38]", "{\"a\": 1e-50, \"b\": 1e39}" + }) + { + CAPTURE(text) + const bool float_accepted = float_json::accept(text); + for (const bool sentinel : + { + true, false + }) + { + const built f = build(text, false, false, sentinel); + CHECK(f.ok == float_accepted); + if (!f.ok) + { + CHECK(f.failure.code == nlohmann::detail::view::error_code::number_overflow); + } + CHECK(build(text, false, false, sentinel).ok == json::accept(text)); + } + } + } + + SECTION("malformed input") + { + for (const char* text : + { + "", " ", "[", "]", "{", "}", "[1,]", "{\"a\":1,}", "[1 2]", "{\"a\" 1}", "{1:2}", "tru", "nul", "fals", "truex", "-", "01", "1.", ".5", "1e", "1e+", + "\"", "\"abc", "\"\\x\"", "\"\\u12\"", "\"\\u12G4\"", "\"\\ud800\"", "\"\\udc00\"", "\"\\ud800\\u0041\"", "\"\x01\"", "\"\xff\"", "\"\xc3\"", // NOLINT(modernize-raw-string-literal) + "\"\xe0\x80\x80\"", "\"\xed\xa0\x80\"", "[1]x", "[1] [2]", "/", "/*", "/* */ 1", "// c\n1", "1 // c", "[1,/*c*/2]", "[1,2,]" + }) + { + check_same(text); + } + } + + SECTION("NUL, BOM, and whitespace") + { + // a NUL inside a string is a control character, as for json::parse + // (where a NUL ends the input, it does so only between values) + for (const bool sentinel : + { + true, false + }) + { + const built b = build(std::string("[\"ab\0cd\"]", 9), false, false, sentinel); + CHECK(!b.ok); + CHECK(b.failure.code == nlohmann::detail::view::error_code::string_control_character); + CHECK(b.failure.offset == 4); + } + check_same(std::string("[1]\0garbage", 11)); + check_same(std::string("[1\0]", 4)); + check_same(std::string("[1, // c\0\n2]", 12)); + check_same(std::string("[1, /* c\0 */ 2]", 15)); + check_same("\xEF\xBB\xBF[1]"); + check_same("\xEF\xBB[1]"); + check_same(" \t\r\n 7 \n"); + for (const char* text : + {"[1]\r", "[1]\n", "[1]\r\n", "[1,\r2]", "[1,\r\n2]", "7\r", "\"x\"\r", "{\"a\":\r\n1}\r", "[\n 1,\n 2\n]", "{\n \"a\": [\n 1\n ]\n}" + }) + { + check_same(text); + } + } + + SECTION("deep nesting") + { + // the open containers beyond 64 levels live on the heap + for (const std::size_t depth : + { + 63u, 64u, 65u, 1000u, 100000u + }) + { + const std::string arrays = std::string(depth, '[') + std::string(depth, ']'); + const built b = build(arrays, false, false, false); + REQUIRE(b.ok); + CHECK(b.data->tape_size == depth); + CHECK(b.data->tape[0].next == depth); + std::string objects; + for (std::size_t i = 0; i < depth; ++i) + { + objects += "{\"a\":"; + } + objects += '1' + std::string(depth, '}'); + const built o = build(objects, false, false, false); + REQUIRE(o.ok); + CHECK(o.data->tape_size == (2 * depth) + 1); + CHECK(!build(std::string(depth, '[') + std::string(depth - 1, ']'), false, false, false).ok); + } + } + + SECTION("generated documents and damaged copies") + { + generator g; + for (int i = 0; i < 3000; ++i) + { + std::string text; + g.value(text, 0); + check_same(text); + // damage: flip one byte, or cut the text + std::string damaged = text; + const auto at = static_cast(g.r(static_cast(damaged.size()))); + static const char replacements[] = {'x', '"', '\\', ',', ':', ']', '}', '[', '{', '1', '-', '.', 'e', '\0', '\n', '/'}; // NOLINT(cppcoreguidelines-avoid-c-arrays,hicpp-avoid-c-arrays,modernize-avoid-c-arrays) + damaged[at] = replacements[g.r(16)]; + check_same(damaged); + check_same(text.substr(0, at)); + } + } + + SECTION("test files") + { + for (const char* name : + { + "/json.org/1.json", "/json.org/2.json", "/json.org/3.json", "/json.org/4.json", "/json.org/5.json", + "/json_testsuite/sample.json", "/nativejson-benchmark/canada.json", "/nativejson-benchmark/citm_catalog.json", + "/nativejson-benchmark/twitter.json", "/json_tests/pass1.json", "/json_tests/pass2.json", "/json_tests/pass3.json" + }) + { + CAPTURE(name) + std::ifstream f(std::string(TEST_DATA_DIRECTORY) + name, std::ios::binary); + std::stringstream ss; + ss << f.rdbuf(); + const std::string text = ss.str(); + REQUIRE(!text.empty()); + const built b = build(text, false, false, true); + REQUIRE(b.ok); + CHECK(value_of(b) == json::parse(text)); + } + } +} diff --git a/tests/src/unit-json_view_macros.cpp b/tests/src/unit-json_view_macros.cpp new file mode 100644 index 000000000..2cc8b7744 --- /dev/null +++ b/tests/src/unit-json_view_macros.cpp @@ -0,0 +1,47 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +#include "doctest_compatibility.h" + +// All other tests keep the library's macros (JSON_TEST_KEEP_MACROS). This one +// includes json_view.hpp as users do, so that json.hpp undefines its macros +// (JSON_HAS_CPP_17, JSON_STRICT_NUL_HANDLING, ...) before the view is compiled. +#undef JSON_TEST_KEEP_MACROS +#include + +#include +#include + +#if (defined(__cplusplus) && __cplusplus >= 201703L) || (defined(_MSVC_LANG) && _MSVC_LANG >= 201703L) + #include + #define JSON_VIEW_TEST_HAS_STRING_VIEW 1 +#else + #define JSON_VIEW_TEST_HAS_STRING_VIEW 0 +#endif + +// the view's own macros do not leak +#if defined(NLOHMANN_VIEW_LIKELY) || defined(NLOHMANN_VIEW_UNLIKELY) || defined(NLOHMANN_VIEW_ALWAYS_INLINE) || defined(NLOHMANN_VIEW_NOINLINE) \ + || defined(NLOHMANN_VIEW_NODISCARD) || defined(NLOHMANN_VIEW_THROW) || defined(NLOHMANN_VIEW_HAS_CPP_17) || defined(NLOHMANN_VIEW_LITTLE_ENDIAN) \ + || defined(NLOHMANN_VIEW_REPEAT16) + #error "json_view.hpp leaks a macro" +#endif + +TEST_CASE("json_view without the library's macros") +{ + // (this file also gets C++17 builds: it mentions JSON_HAS_CPP_17) +#if JSON_VIEW_TEST_HAS_STRING_VIEW + CHECK(std::is_same::value); +#endif + const std::string text = "[1, 2.5, \"x\"]"; + const nlohmann::json_document d = nlohmann::json_document::parse(text); + CHECK(d.root().size() == 3); + CHECK(d.root().materialize() == nlohmann::json::parse(text)); + // the NUL handling of the library's configuration + const std::string with_nul("[1]\0x", 5); + CHECK(nlohmann::json_document::accept(with_nul) == nlohmann::json::accept(with_nul)); +}