From 5a936a2ef3e225574dc18fb916998edf580cf93c Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Tue, 29 Sep 2026 02:08:38 +0200 Subject: [PATCH] Document the images of json_documents Add API pages for basic_json_document::save() and load(), including the image_check enumeration and its three levels. Add examples that show caching a parsed document as an image, loading it without parsing, the difference between a borrowed and an owned image, and a full check rejecting a damaged image that a bounds check still reads safely. Add an "Images" section to the json_view feature page, register the new pages in mkdocs.yml and docSet.sql, group basic_json_document's member list by parsing/access/images/edits, and document parse_error.116 and type_error.320 on the exceptions page, extending out_of_range.416 for images. Signed-off-by: Niels Lohmann --- docs/docset/docSet.sql | 2 + .../docs/api/basic_json_document/index.md | 14 ++ .../docs/api/basic_json_document/load.md | 169 ++++++++++++++++++ .../docs/api/basic_json_document/parse.md | 1 + .../docs/api/basic_json_document/read.md | 1 + .../docs/api/basic_json_document/save.md | 103 +++++++++++ .../examples/basic_json_document__load.cpp | 52 ++++++ .../examples/basic_json_document__load.output | 5 + .../examples/basic_json_document__save.cpp | 30 ++++ .../examples/basic_json_document__save.output | 3 + docs/mkdocs/docs/features/index.md | 3 +- docs/mkdocs/docs/features/json_view.md | 52 ++++++ docs/mkdocs/docs/home/exceptions.md | 44 ++++- docs/mkdocs/mkdocs.yml | 2 + 14 files changed, 479 insertions(+), 2 deletions(-) create mode 100644 docs/mkdocs/docs/api/basic_json_document/load.md create mode 100644 docs/mkdocs/docs/api/basic_json_document/save.md create mode 100644 docs/mkdocs/docs/examples/basic_json_document__load.cpp create mode 100644 docs/mkdocs/docs/examples/basic_json_document__load.output create mode 100644 docs/mkdocs/docs/examples/basic_json_document__save.cpp create mode 100644 docs/mkdocs/docs/examples/basic_json_document__save.output diff --git a/docs/docset/docSet.sql b/docs/docset/docSet.sql index 42b92757c..cf7cee5a3 100644 --- a/docs/docset/docSet.sql +++ b/docs/docset/docSet.sql @@ -134,6 +134,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::accept', INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::erase', 'Method', 'api/basic_json_document/erase/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::insert', 'Method', 'api/basic_json_document/insert/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::is_discarded', 'Method', 'api/basic_json_document/is_discarded/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::load', 'Function', 'api/basic_json_document/load/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::memory_usage', 'Method', 'api/basic_json_document/memory_usage/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::node_count', 'Method', 'api/basic_json_document/node_count/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::owns_source', 'Method', 'api/basic_json_document/owns_source/index.html'); @@ -142,6 +143,7 @@ INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::parse_co INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::push_back', 'Method', 'api/basic_json_document/push_back/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::read', 'Method', 'api/basic_json_document/read/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::root', 'Method', 'api/basic_json_document/root/index.html'); +INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::save', 'Method', 'api/basic_json_document/save/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::set', 'Method', 'api/basic_json_document/set/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::shrink_to_fit', 'Method', 'api/basic_json_document/shrink_to_fit/index.html'); INSERT INTO searchIndex(name, type, path) VALUES ('basic_json_document::source', 'Method', 'api/basic_json_document/source/index.html'); diff --git a/docs/mkdocs/docs/api/basic_json_document/index.md b/docs/mkdocs/docs/api/basic_json_document/index.md index 534b63755..539437616 100644 --- a/docs/mkdocs/docs/api/basic_json_document/index.md +++ b/docs/mkdocs/docs/api/basic_json_document/index.md @@ -52,10 +52,16 @@ bookkeeping edits need, and calling any of them on one fails to compile (`#!cpp ## Member functions - [(constructor)](basic_json_document.md) + +### Parsing + - [**parse**](parse.md) (_static_) - deserialize from a compatible input, borrowing or owning it as appropriate - [**parse_copy**](parse_copy.md) (_static_) - deserialize a copy of a compatible input - [**accept**](accept.md) (_static_) - check whether the input is valid JSON - [**read**](read.md) - (re-)parse into this document, reusing its memory + +### Access + - [**root**](root.md) - the view of the root value - [**is_discarded**](is_discarded.md) - return whether the last parse failed - [**source**](source.md) - the parsed text @@ -63,6 +69,14 @@ bookkeeping edits need, and calling any of them on one fails to compile (`#!cpp - [**node_count**](node_count.md) - the number of index entries (values plus object keys) - [**memory_usage**](memory_usage.md) - the number of bytes held by the document - [**shrink_to_fit**](shrink_to_fit.md) - release unused index capacity + +### Images + +- [**save**](save.md) - the document as an image that `load()` reads without parsing +- [**load**](load.md) (_static_) - read an image written by `save()` + +### Edits + - [**set**](set.md) - replace a value, or set an object member, an array element, or the value a JSON pointer refers to (`#!cpp Editable` documents only) - [**push_back**](push_back.md) - append to an array (`#!cpp Editable` documents only) diff --git a/docs/mkdocs/docs/api/basic_json_document/load.md b/docs/mkdocs/docs/api/basic_json_document/load.md new file mode 100644 index 000000000..2f9d72692 --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/load.md @@ -0,0 +1,169 @@ +# nlohmann::basic_json_document::load + +```cpp +// (1) +static basic_json_document load(const std::uint8_t* image, std::size_t size, + const image_check check = image_check::full); + +// (2) +static basic_json_document load(const std::vector& image, + const image_check check = image_check::full); + +// (3) +static basic_json_document load(std::vector&& image, + const image_check check = image_check::full); +``` + +1. Reads an image [`save()`](save.md) wrote, from a pointer and a byte count. The image is **borrowed**: `image` + must stay alive and unchanged for as long as the returned document, and any view taken from it, is used. +2. Reads an image from a `#!cpp std::vector`. Also **borrowed** -- equivalent to overload 1 called with + `#!cpp image.data()` and `#!cpp image.size()`. +3. Reads an image, keeping the vector instead of copying it: `image` is moved into the document (no copy), which + then owns it for as long as it needs the text and the decoded strings. [`owns_source()`](owns_source.md) is + `#!cpp true` afterward. + +In every overload, the node index is copied into storage the document itself owns -- so that it is properly aligned, +and, for an [editable](index.md#edits) document, can be edited -- while the text and the decoded strings stay in +`image`. The hash indexes [large objects](../../features/json_view.md) use for lookup are rebuilt, exactly as after +parsing. + +## Parameters + +`image` (in) +: the image [`save()`](save.md) wrote (overloads 1 and 2), or one to take ownership of (overload 3) + +`size` (in) +: the number of bytes at `image` (overload 1) + +`check` (in) +: how thoroughly to validate `image` before trusting it; see [`image_check`](#image_check) below (optional, + `#!cpp image_check::full` by default) + +## Return value + +The document read from the image. + +## Exception safety + +Overloads 1 and 2 give the strong guarantee: `image` is only read, never written, so a thrown exception leaves the +caller's buffer untouched. + +Overload 3 moves `image` into the document *before* validating it, so that a good image is kept without a copy. If +loading then fails, the partially built document -- and the vector now inside it -- is discarded along with the +exception, and `image` itself is left **empty**, not restored to what was passed in. Move a copy in instead, or +validate with overload 2 first, if the original vector must survive a failed load. + +## Exceptions + +On a big-endian target, throws [`type_error.320`](../../home/exceptions.md#jsonexceptiontype_error320) -- the same +exception [`save()`](save.md#exceptions) throws there, since the image format is little-endian only. + +Otherwise throws [`parse_error.116`](../../home/exceptions.md#jsonexceptionparse_error116) if `image` is not one +`save()` could have written, or fails the requested `check`: + +| message | when | +|------------------------|------------------------------------------------------------------------------------------------------| +| `too short` | `image` is `#!cpp nullptr`, or `size` is smaller than the 64-byte header | +| `unknown format` | the header's magic bytes or version do not match, or a reserved header field is not zero | +| `sizes out of range` | the node count, text size, or decoded-string size the header describes does not fit `size`, or the `#!cpp '\0'` after the text or after the decoded strings is missing | +| `the check failed` | `check` is not `#!cpp image_check::none`, and the image fails it -- see [`image_check`](#image_check) | + +!!! failure "Example messages" + + ``` + [json.exception.parse_error.116] parse error: invalid json_document image: too short + ``` + ``` + [json.exception.parse_error.116] parse error: invalid json_document image: unknown format + ``` + ``` + [json.exception.parse_error.116] parse error: invalid json_document image: sizes out of range + ``` + ``` + [json.exception.parse_error.116] parse error: invalid json_document image: the check failed + ``` + +## Complexity + +Linear in the number of nodes, which are always copied into the document. With `#!cpp check == image_check::full`, +additionally linear in the combined length of the text and the decoded strings; `#!cpp image_check::bounds` and +`#!cpp image_check::none` do not read them. + +## `image_check` + +```cpp +using image_check = detail::view::image_check; + +enum class image_check +{ + full, + bounds, + none +}; +``` + +How thoroughly `load()` validates `image` before trusting it. + +| value | checks | guarantees | +|----------|--------------------------------------------------------------------------------------------------------------|------------| +| `full` | everything the parser itself guarantees: structure and bounds; that every string is valid UTF-8 (and, for a string still in the source text, that it contains no quote, backslash, or control character); and that every number token is well-formed and matches the value stored for it | reading and serializing a checked image is safe and always produces valid JSON, exactly as for a parsed document | +| `bounds` | structure and bounds only -- that every offset and count in the node index stays inside the image | reading and serializing stay memory-safe, but a crafted image can hold strings that are not valid UTF-8 or that serialize to invalid JSON ([`dump()`](../basic_json_view/dump.md) writes them unchanged or throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316)), and numbers whose values differ from their text | +| `none` | nothing | images from a trusted source only -- reading a damaged image is undefined behavior | + +`full` is the default and the right choice for an image from anything you do not fully control -- a file, a cache +shared with other processes, a peer on the network. `bounds` skips scanning the text and the decoded strings, so it +fits a cache your own process just wrote and reads straight back, where damage would mean a bug or a hardware fault +rather than adversarial input; it still cannot crash or read out of bounds. `none` skips validation entirely and +should only be used for an image you trust as much as your own memory. + +## Notes + +**Lifetime.** Overloads 1 and 2 borrow `image`: it must stay alive and byte-for-byte unchanged for as long as the +returned document, and any [view](../basic_json_view/index.md) taken from it, is used -- exactly like a document +[`parse()`](parse.md) borrowed its input for. Overload 3 avoids this by keeping the vector itself; see +[`owns_source`](owns_source.md). + +!!! warning "Experimental" + + The image format is versioned but not yet stable, and may change in an incompatible way before it is declared + stable; `load()` already rejects an image written by a different format version with `parse_error.116` + ("unknown format"). Use images to cache a document within one build of the library, or to hand one to another + process running the *same* build on the *same* (little-endian) machine -- not as a long-term storage format. + +**What `image_check::bounds` does not guarantee.** A bounds-checked image can never make `load()`, +[`root()`](root.md), element access, or [`materialize()`](../basic_json_view/materialize.md) read outside the image, +so those stay safe on a damaged one. It does *not* guarantee that the image describes valid JSON: a string +that a `full` check would have rejected can make [`dump()`](../basic_json_view/dump.md) write invalid UTF-8 or invalid +JSON, or throw `type_error.316`, and a number can read back with a value that does not match how it is spelled. Reserve `bounds` for images you already trust to be well-formed, and use it +only to skip the extra scan. + +## Examples + +??? example "Caching a document, ownership, and a rejected image" + + The example below saves a parsed document as an image, checks that `load()` reproduces the original + [`dump()`](../basic_json_view/dump.md) without parsing, and shows the difference between + `load(std::move(image))` (owned) and `load(image)` (borrowed). It then damages one byte of the image and shows + `image_check::full` rejecting it with `parse_error.116`, while `image_check::bounds` -- meant for a cache the + process already trusts -- still reads it without going out of bounds. + + ```cpp + --8<-- "examples/basic_json_document__load.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__load.output" + ``` + +## See also + +- [save](save.md) - write the document as an image +- [owns_source](owns_source.md) - return whether the document holds its own copy of the text +- [parse](parse.md) - deserialize from JSON text instead of an image +- [Images](../../features/json_view.md#images) - why and when to use images + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/api/basic_json_document/parse.md b/docs/mkdocs/docs/api/basic_json_document/parse.md index a3c44f4e0..88a4fc20c 100644 --- a/docs/mkdocs/docs/api/basic_json_document/parse.md +++ b/docs/mkdocs/docs/api/basic_json_document/parse.md @@ -132,6 +132,7 @@ integer type becomes a floating-point value. - [accept](accept.md) - check whether the input is valid JSON - [read](read.md) - (re-)parse into this document, reusing its memory - [owns_source](owns_source.md) - return whether the document holds its own copy of the text +- [load](load.md) - read a document from an image instead of parsing JSON text - [`BasicJsonType::parse`](../basic_json/parse.md) - the corresponding function of `basic_json` ## Version history diff --git a/docs/mkdocs/docs/api/basic_json_document/read.md b/docs/mkdocs/docs/api/basic_json_document/read.md index 2267e46bd..fd6c45657 100644 --- a/docs/mkdocs/docs/api/basic_json_document/read.md +++ b/docs/mkdocs/docs/api/basic_json_document/read.md @@ -70,6 +70,7 @@ own on the next, since ownership is decided freshly each time. - [parse](parse.md) - deserialize from a compatible input - [root](root.md) - the view of the root value +- [load](load.md) - read a document from an image instead of parsing JSON text ## Version history diff --git a/docs/mkdocs/docs/api/basic_json_document/save.md b/docs/mkdocs/docs/api/basic_json_document/save.md new file mode 100644 index 000000000..a80927c0c --- /dev/null +++ b/docs/mkdocs/docs/api/basic_json_document/save.md @@ -0,0 +1,103 @@ +# nlohmann::basic_json_document::save + +```cpp +std::vector save() const; +``` + +Writes the document as an *image*: a byte buffer that [`load`](load.md) reads back without parsing. The image holds +the node index, the source text (plus, for an edited document, the number tokens edits wrote), and the decoded +strings (plus the strings edits wrote) -- everything [`root()`](root.md) needs, with nothing left to parse. + +An edited document is written in its *current* state, with its values in document order, the way the library's own +parser would have produced them for that JSON text: a member [`set`](set.md) added goes at the end, an +[`erase`](erase.md)d member leaves no trace, and a float that is not finite (NaN or positive/negative infinity) +becomes null, the same substitution [`dump()`](../basic_json_view/dump.md) makes. The same document always saves to +the same bytes -- also across `BasicJsonType` and `#!cpp Editable`, since the image reflects document order and +values only, not which specialization produced them. + +## Return value + +The image, as a `#!cpp std::vector`. Pass it, or a pointer to its data together with its size, to +[`load`](load.md) to read the document back. + +## Exception safety + +Strong guarantee: `save()` does not modify `#!cpp *this` (it is `#!cpp const`), so if it throws, the document is left +exactly as it was, and the partially built image is discarded with the exception. + +## Exceptions + +Throws [`type_error.320`](../../home/exceptions.md#jsonexceptiontype_error320) if the document is +[discarded](is_discarded.md) -- a default-constructed document, or one a failed [`parse()`](parse.md)/ +[`read()`](read.md) with `allow_exceptions == false` left discarded. + +On a big-endian target, throws `type_error.320` with a different message instead: the image format is little-endian +only (see [Notes](#notes)). + +Throws [`out_of_range.416`](../../home/exceptions.md#jsonexceptionout_of_range416) if the node count, the text, or +the decoded strings of the image would individually reach 4 GiB -- the same 32-bit offsets +[`parse()`](parse.md#exceptions) and, for edits, [`set`](set.md)/[`push_back`](push_back.md) are already limited to. + +!!! failure "Example messages" + + ``` + [json.exception.type_error.320] cannot save a discarded json_document + ``` + ``` + [json.exception.type_error.320] json_document images need a little-endian target + ``` + ``` + [json.exception.out_of_range.416] images of 4 GiB or more are not supported by json_document + ``` + +## Complexity + +Linear in the size of the document: the number of nodes, plus the length of the text and the decoded strings that end +up in the image. + +## Notes + +**Format.** The image begins with a 64-byte header (the magic bytes `#!cpp "NJVI"`, a version number, the node count, +and the sizes of the text and the decoded strings, all little-endian), followed by the nodes (16 bytes each), the +text and a `#!cpp '\0'`, and the decoded strings and a `#!cpp '\0'`. [`load`](load.md) checks the header, and the +sizes it describes, before reading anything else -- see [`load`'s Exceptions](load.md#exceptions). + +!!! warning "Experimental" + + The image format is versioned but not yet stable: it may change in an incompatible way before it is declared + stable. Use images to cache a document within one build of the library, or to hand one to another process running + the *same* build on the *same* (little-endian) machine -- not as a long-term storage format. Keep the original + JSON text if you need to read a saved document back with a future library version. + +**Little-endian only.** The image is written as raw little-endian bytes, with no byte-swapping. `save()` (and +[`load`](load.md)) throw `type_error.320` on a big-endian target rather than silently produce bytes a big-endian +reader could not interpret correctly. + +## Examples + +??? example "Caching a parsed document as an image" + + The example below saves a parsed configuration as an image -- the way a service might cache one to answer later + requests without parsing the text again -- and confirms that loading it back gives exactly the same result as + parsing did, and that saving is deterministic. + + ```cpp + --8<-- "examples/basic_json_document__save.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__save.output" + ``` + +## See also + +- [load](load.md) - read an image written by `save()` +- [owns_source](owns_source.md) - return whether the document holds its own copy of the text +- [`basic_json_view::dump`](../basic_json_view/dump.md) - serialize the document to JSON text instead of an image +- [Images](../../features/json_view.md#images) - why and when to use images + +## Version history + +- Added in version 3.13.0. diff --git a/docs/mkdocs/docs/examples/basic_json_document__load.cpp b/docs/mkdocs/docs/examples/basic_json_document__load.cpp new file mode 100644 index 000000000..52a085a10 --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__load.cpp @@ -0,0 +1,52 @@ +#include +#include +#include +#include +#include + +using json = nlohmann::json; +using json_document = nlohmann::json_document; +using image_check = json_document::image_check; + +int main() +{ + std::cout << std::boolalpha; + + // the image of a parsed document -- as if read back from a cache file or + // received from another process running the same build of the library + const std::string text = R"({"name": "cache", "note": "caf\u00e9", "replicas": ["db2", "db3"]})"; + const json_document parsed = json_document::parse(text); + const std::vector image = parsed.save(); + + // (1)/(2) load() needs no parsing, yet dumps exactly what parsing did + const json_document borrowed = json_document::load(image); + std::cout << (borrowed.root().dump() == parsed.root().dump()) << '\n'; + std::cout << borrowed.owns_source() << '\n'; // borrowed: still points into `image` + + // (3) load(std::move(image)) keeps the vector instead of copying it + std::vector to_move = image; + const json_document owned = json_document::load(std::move(to_move)); + std::cout << owned.owns_source() << '\n'; + + // a damaged image -- the last byte of the decoded string "note" holds + // (an escape sequence, so it was unescaped into the document's own + // buffer), flipped, as storage or transport corruption might do + std::vector damaged = image; + damaged[damaged.size() - 2] = 0xFF; + + // image_check::full inspects strings and numbers, so it catches the damage + try + { + static_cast(json_document::load(damaged, image_check::full)); + } + catch (const json::parse_error& e) + { + std::cout << e.id << '\n'; + } + + // image_check::bounds only checks structure and bounds, so a cache the + // process already trusts loads without the extra scan -- reading a value + // the damage did not touch is still safe + const json_document trusted = json_document::load(damaged, image_check::bounds); + std::cout << trusted.root()["name"].get() << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__load.output b/docs/mkdocs/docs/examples/basic_json_document__load.output new file mode 100644 index 000000000..4dc66a29b --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__load.output @@ -0,0 +1,5 @@ +true +false +true +116 +cache diff --git a/docs/mkdocs/docs/examples/basic_json_document__save.cpp b/docs/mkdocs/docs/examples/basic_json_document__save.cpp new file mode 100644 index 000000000..6a3cc978b --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__save.cpp @@ -0,0 +1,30 @@ +#include +#include +#include +#include +#include + +using json_document = nlohmann::json_document; + +int main() +{ + std::cout << std::boolalpha; + + // a configuration a service parses once and then caches as an image, so + // that later requests can load() it instead of parsing the text again + const std::string text = R"({"name": "cache", "host": "db1", "port": 6379, "replicas": ["db2", "db3"]})"; + const json_document config = json_document::parse(text); + + // save() turns the parsed document into a byte buffer: a 64-byte header, + // the node index, the source text, and the decoded strings + const std::vector image = config.save(); + std::cout << image.size() << '\n'; + + // the same document always saves to the same bytes + std::cout << (image == json_document::parse(text).save()) << '\n'; + + // loading the image back needs no parsing, yet dumps exactly what + // parsing the text produced + const json_document reloaded = json_document::load(image); + std::cout << (reloaded.root().dump() == config.root().dump()) << '\n'; +} diff --git a/docs/mkdocs/docs/examples/basic_json_document__save.output b/docs/mkdocs/docs/examples/basic_json_document__save.output new file mode 100644 index 000000000..5ce0ade9a --- /dev/null +++ b/docs/mkdocs/docs/examples/basic_json_document__save.output @@ -0,0 +1,3 @@ +316 +true +true diff --git a/docs/mkdocs/docs/features/index.md b/docs/mkdocs/docs/features/index.md index e17d0dabe..7affb523d 100644 --- a/docs/mkdocs/docs/features/index.md +++ b/docs/mkdocs/docs/features/index.md @@ -13,7 +13,8 @@ C++ types, and finally serialize it again. [SAX interface](parsing/sax_interface.md), and [error handling](parsing/parse_exceptions.md). - [Zero-copy JSON views](json_view.md) — read a JSON text through a flat index instead of building a `json` tree; strings and numbers stay in the input and are only decoded when needed. - [Editable documents](json_view.md#editing-a-document) can also be modified. + [Editable documents](json_view.md#editing-a-document) can also be modified, and [images](json_view.md#images) load a + parsed document again without parsing it. - [Comments](comments.md) and [trailing commas](trailing_commas.md) — opt-in relaxations of the JSON grammar. ## Accessing and modifying values diff --git a/docs/mkdocs/docs/features/json_view.md b/docs/mkdocs/docs/features/json_view.md index 8a275db4a..efef2bda2 100644 --- a/docs/mkdocs/docs/features/json_view.md +++ b/docs/mkdocs/docs/features/json_view.md @@ -250,6 +250,57 @@ edits. See [`basic_json_document`'s Edits](../api/basic_json_document/index.md#e throws (the *basic* guarantee, not the strong one `dump()` and the read-only functions provide). How edits are kept in the index is described in the [architecture overview](../home/architecture.md#node-index-of-json-views). +## Images + +[`save()`](../api/basic_json_document/save.md) writes a document as an *image*: a byte buffer that +[`load()`](../api/basic_json_document/load.md) reads back into a document without parsing -- no lexing, no building +the node index, nothing but copying the nodes and pointing the text and the decoded strings at the image. Where +[`parse_copy()`](../api/basic_json_document/parse_copy.md) still has to scan the whole input, +[`load()`](../api/basic_json_document/load.md) turns that scan into a copy of the node index alone. + +**Why.** A document that is parsed once and then read many times -- a configuration loaded at startup, a template +rendered on every request, a large reference dataset a worker process needs in memory -- pays for parsing once but +can amortize [`save()`](../api/basic_json_document/save.md)'s cost across every later load. That makes images useful +for a cache: save a document the first time it is parsed (to a file, a shared-memory segment, an in-process cache), +and [`load()`](../api/basic_json_document/load.md) it on every later use instead of parsing the source text again. +They are just as useful for handing a parsed document to another process (or a forked worker) running the same build +of the library, since [`load()`](../api/basic_json_document/load.md) turns the transfer into a copy of the node index +plus pointers into the received bytes, not a re-parse. + +**Choosing a check.** [`load()`](../api/basic_json_document/load.md) takes an +[`image_check`](../api/basic_json_document/load.md#image_check) that trades validation against speed: +`image_check::full` (the default) checks everything the parser itself guarantees, so a checked image is exactly as +safe to read and serialize as a freshly parsed document -- the right choice whenever the image did not come straight +from this process's own [`save()`](../api/basic_json_document/save.md), such as a file or a network peer. +`image_check::bounds` only checks structure and bounds -- cheaper, since it skips scanning the text and the decoded +strings -- and fits a cache the process trusts, one it wrote and reads back itself. `image_check::none` skips +validation entirely, for an image trusted as much as the process's own memory. See +[`load()`'s Notes](../api/basic_json_document/load.md#notes) for exactly what each level does and does not guarantee. + +??? example "Example: cache a parsed configuration as an image" + + ```cpp + --8<-- "examples/basic_json_document__save.cpp" + ``` + + Output: + + ```json + --8<-- "examples/basic_json_document__save.output" + ``` + +!!! warning "Experimental" + + The image format is versioned but not yet stable, and may change in an incompatible way before it is declared + stable. It is little-endian only, and tied to the library build that wrote it -- use it to cache a document or to + hand one to another process running the *same* build, not as a long-term storage format; keep the original JSON + text if a saved document needs to be readable by a future library version. + +The idea of a document you can read without parsing comes from zero-copy formats such as +[FlatBuffers](https://github.com/google/flatbuffers) and [YaFF](https://github.com/yandex/yaff); the check +[`load()`](../api/basic_json_document/load.md) runs follows the idea of FlatBuffers' Verifier. No code is taken from +either. + ## Choosing between `json`, `ordered_json`, the SAX interface, and `json_view` | | [`json`](../api/json.md) / [`ordered_json`](../api/ordered_json.md) | [SAX interface](parsing/sax_interface.md) | [`json_document`](../api/json_document.md) / [`json_view`](../api/json_view.md) | [`json_editable_document`](../api/json_editable_document.md) / [`json_editable_view`](../api/json_editable_view.md) | @@ -258,6 +309,7 @@ the index is described in the [architecture overview](../home/architecture.md#no | **Mutability** | freely mutable | not applicable (a one-shot event stream) | read-only | [`set`](../api/basic_json_document/set.md)/[`push_back`](../api/basic_json_document/push_back.md)/[`insert`](../api/basic_json_document/insert.md)/[`erase`](../api/basic_json_document/erase.md) edit in place; the source text is never rewritten | | **What you get** | a full tree you can read, write, and keep as long as you like | a sequence of callbacks; whatever your handler builds from them | a flat index plus, on demand, [`materialize()`](../api/basic_json_view/materialize.md)d `json`/`ordered_json` values for the parts you actually use | the same, plus [`dump()`](../api/basic_json_view/dump.md) of an edited document that keeps the member order and, with [`number_format::source`](../api/basic_json_view/number_format.md), the spelling of every untouched number | | **Typical use** | general-purpose JSON handling: config, request/response bodies you build or modify, anything you hold onto | validating or projecting a text into your own data structure without ever holding the whole thing as JSON | large or high-volume input where you only need part of it, or need it repeatedly, and can keep the source text (or a copy) alive for as long as the document lives | a document you read, patch a few fields of, and write back -- a configuration file, for instance -- where the rest of it should come back exactly as it was | +| **Caching/reload** | not applicable -- re-parse, or roll your own serialization | not applicable | [`save()`](../api/basic_json_document/save.md)/[`load()`](../api/basic_json_document/load.md): cache the parsed index as an image and reload it without parsing | same, saving the document's current -- possibly edited -- state | ## Version history diff --git a/docs/mkdocs/docs/home/exceptions.md b/docs/mkdocs/docs/home/exceptions.md index 5ba54acfc..4a73f09d3 100644 --- a/docs/mkdocs/docs/home/exceptions.md +++ b/docs/mkdocs/docs/home/exceptions.md @@ -388,6 +388,23 @@ A UBJSON high-precision number could not be parsed. [json.exception.parse_error.115] parse error at byte 5: syntax error while parsing UBJSON high-precision number: invalid number text: 1A ``` +### json.exception.parse_error.116 + +[`basic_json_document::load()`](../api/basic_json_document/load.md) rejected an +[image](../features/json_view.md#images): either the bytes are not one [`save()`](../api/basic_json_document/save.md) +could have written (too short, an unknown magic number or format version, or sizes that do not fit the buffer), or +they are, but fail the requested [`image_check`](../api/basic_json_document/load.md#image_check). + +!!! failure "Example message" + + ``` + [json.exception.parse_error.116] parse error: invalid json_document image: the check failed + ``` + +!!! note + + This exception was added in version 3.13.0, together with [images](../features/json_view.md#images). + ## Iterator errors This exception is thrown if iterators passed to a library function do not match @@ -800,6 +817,26 @@ from JSON text. This exception was added in version 3.13.0, together with editable [`json_document`s](../features/json_view.md). +### json.exception.type_error.320 + +[`basic_json_document::save()`](../api/basic_json_document/save.md) cannot write an +[image](../features/json_view.md#images) of a [discarded](../api/basic_json_document/is_discarded.md) document. +[`save()`](../api/basic_json_document/save.md) and [`load()`](../api/basic_json_document/load.md) also throw this +exception on a big-endian target, since the image format is little-endian only. + +!!! failure "Example messages" + + ``` + [json.exception.type_error.320] cannot save a discarded json_document + ``` + ``` + [json.exception.type_error.320] json_document images need a little-endian target + ``` + +!!! note + + This exception was added in version 3.13.0, together with [images](../features/json_view.md#images). + ## Out of range This exception is thrown in case a library function is called on an input parameter that exceeds the expected range, for instance, in the case of array indices or nonexisting object keys. @@ -1027,7 +1064,9 @@ MessagePack's ext type and BSON's binary subtype are each stored in a single byt so they do not support an input of 4 GiB or more. The same 32-bit limit applies to an **editable** document's own storage: [`set`](../api/basic_json_document/set.md) and [`push_back`](../api/basic_json_document/push_back.md) throw this exception once the strings and number tokens written by edits reach 4 GiB in total, or once more than -4294967295 arrays/objects have had an element set or appended to them. +4294967295 arrays/objects have had an element set or appended to them. The same limit applies to an +[image](../features/json_view.md#images): [`save()`](../api/basic_json_document/save.md) throws it if the node +count, the text, or the decoded strings it would write would individually reach 4 GiB. !!! failure "Example messages" @@ -1037,6 +1076,9 @@ this exception once the strings and number tokens written by edits reach 4 GiB i ``` [json.exception.out_of_range.416] edits of 4 GiB or more are not supported by json_document ``` + ``` + [json.exception.out_of_range.416] images of 4 GiB or more are not supported by json_document + ``` !!! note diff --git a/docs/mkdocs/mkdocs.yml b/docs/mkdocs/mkdocs.yml index 462fddb34..cbf7bf1fd 100644 --- a/docs/mkdocs/mkdocs.yml +++ b/docs/mkdocs/mkdocs.yml @@ -236,6 +236,7 @@ nav: - 'erase': api/basic_json_document/erase.md - 'insert': api/basic_json_document/insert.md - 'is_discarded': api/basic_json_document/is_discarded.md + - 'load': api/basic_json_document/load.md - 'memory_usage': api/basic_json_document/memory_usage.md - 'node_count': api/basic_json_document/node_count.md - 'owns_source': api/basic_json_document/owns_source.md @@ -244,6 +245,7 @@ nav: - 'push_back': api/basic_json_document/push_back.md - 'read': api/basic_json_document/read.md - 'root': api/basic_json_document/root.md + - 'save': api/basic_json_document/save.md - 'set': api/basic_json_document/set.md - 'shrink_to_fit': api/basic_json_document/shrink_to_fit.md - 'source': api/basic_json_document/source.md