From 26b7bb2ff4790954083f9160b936b6440a986b61 Mon Sep 17 00:00:00 2001 From: Niels Lohmann Date: Mon, 28 Sep 2026 23:20:23 +0200 Subject: [PATCH] Compare json_view with yyjson, simdjson, and Boost.JSON tests/benchmarks/json_view/ holds the comparison with other libraries, which is not built by CMake or run by CI: - bench_view.cpp: parse, traverse, select, and dump of twitter, citm_catalog, canada, jeopardy, a single tweet, and a JSON-RPC request, with json_view, yyjson, simdjson (DOM and On-Demand), Boost.JSON, and json::parse; all engines must agree on every document before anything is timed, and run interleaved in every round - bench_corpus.cpp: parse, traverse, and dump of any list of files - compare.py: builds both against include/ with the libraries of the system (or pinned downloads), runs them, and writes the results with what is needed to reproduce them (date, commit, CPU, OS, compiler, flags, library versions) to results/-.md and .csv; only the Python 3 standard library is used - README.md: how to run it, what is measured, and which features the engines have, so the numbers can be read correctly Boost.JSON is optional (JSON_VIEW_BENCH_BOOST). Signed-off-by: Niels Lohmann --- .github/labeler.yml | 1 + tests/benchmarks/json_view/.gitignore | 1 + tests/benchmarks/json_view/README.md | 65 ++ tests/benchmarks/json_view/bench_corpus.cpp | 326 +++++++++ tests/benchmarks/json_view/bench_view.cpp | 735 ++++++++++++++++++++ tests/benchmarks/json_view/compare.py | 295 ++++++++ 6 files changed, 1423 insertions(+) create mode 100644 tests/benchmarks/json_view/.gitignore create mode 100644 tests/benchmarks/json_view/README.md create mode 100644 tests/benchmarks/json_view/bench_corpus.cpp create mode 100644 tests/benchmarks/json_view/bench_view.cpp create mode 100755 tests/benchmarks/json_view/compare.py diff --git a/.github/labeler.yml b/.github/labeler.yml index 7884e69fb..a135a10f1 100644 --- a/.github/labeler.yml +++ b/.github/labeler.yml @@ -52,6 +52,7 @@ labels: - "single_include/nlohmann/json_view\\.hpp" - "tests/src/unit-json_view.*" - "tests/src/fuzzer-parse_json_view\\.cpp" + - "tests/benchmarks/json_view/.*" - "tools/amalgamate/config_json_view\\.json" - "docs/mkdocs/docs/features/json_view\\.md" - "docs/mkdocs/docs/api/basic_json_(document|view)/.*" diff --git a/tests/benchmarks/json_view/.gitignore b/tests/benchmarks/json_view/.gitignore new file mode 100644 index 000000000..567609b12 --- /dev/null +++ b/tests/benchmarks/json_view/.gitignore @@ -0,0 +1 @@ +build/ diff --git a/tests/benchmarks/json_view/README.md b/tests/benchmarks/json_view/README.md new file mode 100644 index 000000000..d5e0678c9 --- /dev/null +++ b/tests/benchmarks/json_view/README.md @@ -0,0 +1,65 @@ +# json_view compared with other libraries + +The in-tree benchmarks in [`tests/benchmarks`](../README.md) measure `json_document` against `json::parse` only. The +programs here compare it with [yyjson](https://github.com/ibireme/yyjson), +[simdjson](https://github.com/simdjson/simdjson), and [Boost.JSON](https://github.com/boostorg/json): the question +users ask when they pick a library. They are not built by CMake or run by CI. + +## Reproducing the numbers + +`compare.py` builds both programs against `include/` of this checkout, runs them, and writes the results together with +everything needed to reproduce them to `results/-.md` (and `.csv`): the date, the commit, the CPU, the +OS, the compiler, the flags, and the versions of all libraries. + +```sh +python3 tests/benchmarks/json_view/compare.py --data [--native] [--rounds 30] +``` + +- `--data` is the downloaded [test data](https://github.com/nlohmann/json_test_data), e.g. the `test_files` directory + of a CMake build directory. It needs `nativejson-benchmark/{twitter,citm_catalog,canada}.json` and + `jeopardy/jeopardy.json`. +- The other libraries come from the system: pkg-config, or Homebrew (`brew install yyjson simdjson boost`). With + `--download`, pinned releases are downloaded instead and checked against their SHA-256. Without Boost headers (or + with `--no-boost`), the Boost.JSON columns are skipped, and the results say so. +- `--corpus file...` adds files to the corpus benchmark, e.g. those of + [simdjson-data](https://github.com/simdjson/simdjson-data) or the + [yyjson benchmark](https://github.com/ibireme/yyjson_benchmark). +- Only the Python 3 standard library is used; a C++17 compiler is needed (`CXX` and `CC` are honored). + +For numbers worth publishing, use a quiet machine (see [Getting stable numbers](../README.md#getting-stable-numbers)), +the default 30 rounds or more, and `--native` only if the other libraries were built for the same CPU. + +## What is measured + +`bench_view.cpp` runs four workloads on twitter, citm_catalog, canada, jeopardy, a single tweet (`status`), and a +JSON-RPC request (`rpc`): + +| workload | what it does | +|---|---| +| parse | build and free a document | +| traverse | parse, then visit every value, convert every number, touch every string and key | +| select | parse, then read a few fields per record (e.g. id, user name, and retweet count of each tweet) | +| dump | serialize a parsed document (compact) | + +`bench_corpus.cpp` runs parse, traverse, and dump on any list of files, so that no library is tuned to a handful of +documents. + +Before anything is timed, all engines must accept each document and agree on the traversal: the number of values, the +bytes of all strings and keys, and the sum of all numbers. All engines run interleaved in every round, and the best +round is reported, as time and as a factor of the `json_view` time (below 1 means faster than `json_view`). + +The engines do not all offer the same features, which the numbers should be read with: + +| engine | document | random access | editable | notes | +|---|---|---|---|---| +| `json_view` | immutable index into the text | yes | no | a fresh document per parse; "reused" parses into the same document | +| yyjson | immutable (`yyjson_read`) | yes | via a mutable copy | | +| simdjson DOM | immutable, parser reused | yes | no | | +| simdjson On-Demand | none: forward-only, lazy | no | no | only traverse and select | +| Boost.JSON | owning, mutable DOM | yes | yes | monotonic resource | +| `json::parse` | owning, mutable DOM | yes | yes | | + +## Published results + +Results are only published with the file `compare.py` wrote, which names the machine and the versions; see +`results/`. Numbers from one machine and compiler do not carry over to another: rerun the script. diff --git a/tests/benchmarks/json_view/bench_corpus.cpp b/tests/benchmarks/json_view/bench_corpus.cpp new file mode 100644 index 000000000..7d4f5b2f9 --- /dev/null +++ b/tests/benchmarks/json_view/bench_corpus.cpp @@ -0,0 +1,326 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Corpus benchmark: the read-only workloads of bench_view.cpp on any list of +// JSON files (for example the benchmark sets of simdjson and yyjson). +// +// ./bench_corpus [--rounds N] file... +// +// For every file, all engines must accept it and agree on a traversal (value +// count, string bytes, sum of numbers) before anything is timed. Workloads: +// parse (build and free a document), traverse (visit every value, convert +// every number), dump (compact), and for json_view also dump with the source +// number text. Results go to bench_corpus.csv. +#include + +#if JSON_VIEW_BENCH_BOOST + #include + #include +#endif +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::json_document; +using nlohmann::json_view; + +static volatile double g_sink; + +struct stats +{ + double num = 0; + std::size_t str = 0, nodes = 0; +}; + +static void walk(json_view v, stats& st) +{ + ++st.nodes; + switch (v.type()) + { + case json::value_t::object: + for (auto it = v.begin(); it != v.end(); ++it) + { + st.str += it.key().size(); + walk(*it, st); + } + break; + case json::value_t::array: + for (const json_view e : v) + { + walk(e, st); + } + break; + case json::value_t::string: + st.str += v.get_string().size(); + break; + case json::value_t::number_integer: + st.num += static_cast(v.get()); + break; + case json::value_t::number_unsigned: + st.num += static_cast(v.get()); + break; + case json::value_t::number_float: + st.num += v.get(); + break; + default: + break; + } +} + +static void walk(yyjson_val* v, stats& st) +{ + ++st.nodes; + switch (yyjson_get_type(v)) + { + case YYJSON_TYPE_OBJ: + { + std::size_t idx, max; + yyjson_val* k, * val; + yyjson_obj_foreach(v, idx, max, k, val) + { + st.str += yyjson_get_len(k); + walk(val, st); + } + break; + } + case YYJSON_TYPE_ARR: + { + std::size_t idx, max; + yyjson_val* val; + yyjson_arr_foreach(v, idx, max, val) + { + walk(val, st); + } + break; + } + case YYJSON_TYPE_STR: + st.str += yyjson_get_len(v); + break; + case YYJSON_TYPE_NUM: + st.num += yyjson_is_sint(v) ? static_cast(yyjson_get_sint(v)) : yyjson_is_uint(v) ? static_cast(yyjson_get_uint(v)) : yyjson_get_real(v); + break; + default: + break; + } +} + +static void walk(simdjson::dom::element e, stats& st) +{ + ++st.nodes; + switch (e.type()) + { + case simdjson::dom::element_type::OBJECT: + for (auto f : simdjson::dom::object(e)) + { + st.str += f.key.size(); + walk(f.value, st); + } + break; + case simdjson::dom::element_type::ARRAY: + for (auto c : simdjson::dom::array(e)) + { + walk(c, st); + } + break; + case simdjson::dom::element_type::STRING: + st.str += std::string_view(e).size(); + break; + case simdjson::dom::element_type::INT64: + st.num += static_cast(int64_t(e)); + break; + case simdjson::dom::element_type::UINT64: + st.num += static_cast(uint64_t(e)); + break; + case simdjson::dom::element_type::DOUBLE: + st.num += double(e); + break; + default: + break; + } +} + +#if JSON_VIEW_BENCH_BOOST +static void walk(const boost::json::value& v, stats& st) +{ + ++st.nodes; + switch (v.kind()) + { + case boost::json::kind::object: + for (const auto& kv : v.get_object()) + { + st.str += kv.key().size(); + walk(kv.value(), st); + } + break; + case boost::json::kind::array: + for (const auto& c : v.get_array()) + { + walk(c, st); + } + break; + case boost::json::kind::string: + st.str += v.get_string().size(); + break; + case boost::json::kind::int64: + st.num += static_cast(v.get_int64()); + break; + case boost::json::kind::uint64: + st.num += static_cast(v.get_uint64()); + break; + case boost::json::kind::double_: + st.num += v.get_double(); + break; + default: + break; + } +} +#endif + +static std::string slurp(const std::string& p) +{ + std::ifstream f(p, std::ios::binary); + std::stringstream ss; + ss << f.rdbuf(); + return ss.str(); +} + +static bool same(const stats& a, const stats& b) +{ + return a.nodes == b.nodes && a.str == b.str && (a.num == b.num || std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num)); +} + +int main(int argc, char** argv) +{ + int rounds = 0; // 0: by size + std::vector files; + for (int i = 1; i < argc; ++i) + { + if (std::strcmp(argv[i], "--rounds") == 0 && i + 1 < argc) + { + rounds = std::atoi(argv[++i]); + } + else + { + files.push_back(argv[i]); + } + } + std::FILE* csv = std::fopen("bench_corpus.csv", "w"); + std::fprintf(csv, "file,bytes,workload,engine,ns\n"); + simdjson::dom::parser sj; + for (const auto& path : files) + { + const std::string s = slurp(path); + const std::string name = path.substr(path.rfind('/') + 1); + const simdjson::padded_string ps(s); + + // all engines must agree before timing + stats a, b, c, d; + const json_document doc = json_document::parse(s); + walk(doc.root(), a); + yyjson_doc* y = yyjson_read(s.data(), s.size(), 0); + auto sjr = sj.parse(ps); +#if JSON_VIEW_BENCH_BOOST + boost::json::parse_options opt; + opt.numbers = boost::json::number_precision::precise; + boost::json::monotonic_resource mr0; + const boost::json::value bv = boost::json::parse(s, &mr0, opt); +#endif + if (y == nullptr || sjr.error()) + { + std::printf("%-34s skipped (an engine rejects it)\n", name.c_str()); + yyjson_doc_free(y); + continue; + } + walk(yyjson_doc_get_root(y), b); + walk(sjr.value_unsafe(), c); +#if JSON_VIEW_BENCH_BOOST + walk(bv, d); +#else + d = a; +#endif + yyjson_doc_free(y); + const bool ok = same(a, b) && same(a, c) && same(a, d); + + const int r = rounds > 0 ? rounds : static_cast(std::max(3, std::min(60, 400000000 / (s.size() + 1)))); + struct engine + { + std::string name; + std::function fn; + }; + json_document vd = json_document::parse(s); + yyjson_doc* yd = yyjson_read(s.data(), s.size(), 0); + simdjson::dom::parser sjd; + const simdjson::dom::element se = sjd.parse(ps).value_unsafe(); + const std::vector>> workloads = + { + { + "parse", { + {"json_view", [&] { auto x = json_document::parse(s); g_sink = static_cast(x.node_count()); }}, + {"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast(yyjson_doc_get_val_count(x)); yyjson_doc_free(x); }}, + {"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }}, +#endif + } + }, + { + "traverse", { + {"json_view", [&] { auto x = json_document::parse(s); stats st; walk(x.root(), st); g_sink = st.num; }}, + {"yyjson", [&] { yyjson_doc* x = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(x), st); g_sink = st.num; yyjson_doc_free(x); }}, + {"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }}, +#endif + } + }, + { + "dump", { + {"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast(o.size()); }}, + {"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast(n); std::free(o); }}, + {"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast(o.size()); }}, + {"json_view (source numbers)", [&] { std::string o = vd.root().dump(-1, ' ', false, json_view::number_format::source); g_sink = static_cast(o.size()); }}, + } + }, + }; + std::printf("%-34s %9zu B%s\n", name.c_str(), s.size(), ok ? "" : " [ENGINES DISAGREE]"); + for (const auto& wl : workloads) + { + std::vector best(wl.second.size(), 1e300); + for (int i = 0; i < r; ++i) + { + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + const auto t0 = std::chrono::steady_clock::now(); + wl.second[k].fn(); + best[k] = std::min(best[k], std::chrono::duration(std::chrono::steady_clock::now() - t0).count()); + } + } + std::printf(" %-9s", wl.first.c_str()); + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + std::printf(" %s %.2f GB/s (%.2fx)", wl.second[k].name.c_str(), static_cast(s.size()) / best[k], best[k] / best[0]); + std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]); + } + std::printf("\n"); + std::fflush(stdout); + } + yyjson_doc_free(yd); + } + std::fclose(csv); +} diff --git a/tests/benchmarks/json_view/bench_view.cpp b/tests/benchmarks/json_view/bench_view.cpp new file mode 100644 index 000000000..fc1af65ee --- /dev/null +++ b/tests/benchmarks/json_view/bench_view.cpp @@ -0,0 +1,735 @@ +// __ _____ _____ _____ +// __| | __| | | | JSON for Modern C++ (supporting code) +// | | |__ | | | | | | version 3.12.0 +// |_____|_____|_____|_|___| https://github.com/nlohmann/json +// +// SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +// SPDX-License-Identifier: MIT + +// Same-feature-set benchmark: read-only JSON documents with random access. +// +// json_view nlohmann/json_view.hpp (fresh document per parse / reused) +// yyjson yyjson_read(): immutable document, random access +// simdjson DOM dom::parser (reused, as recommended): immutable, random access +// references (different feature sets): +// simdjson OD On-Demand: forward-only, lazy +// Boost.JSON owning, mutable DOM (monotonic resource) +// json::parse owning, mutable DOM (nlohmann today) +// +// Workloads: parse (build + free), traverse (visit everything, convert every +// number, touch every string and key), select (a few fields per document), +// dump (compact serialization of the parsed document). +// All engines run interleaved in every round; the best round is reported. +#include + +#if JSON_VIEW_BENCH_BOOST + #include + #include +#endif +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +using nlohmann::json; +using nlohmann::json_document; +using nlohmann::json_view; + +static volatile double g_sink; + +struct stats +{ + double num = 0; + std::size_t str = 0, nodes = 0; +}; + +// ---------------- traversal ---------------- + +static void walk(json_view v, stats& st) +{ + ++st.nodes; + switch (v.type()) + { + case json::value_t::object: + for (auto it = v.begin(); it != v.end(); ++it) + { + st.str += it.key().size(); + walk(*it, st); + } + break; + case json::value_t::array: + for (const json_view e : v) + { + walk(e, st); + } + break; + case json::value_t::string: + st.str += v.get_string().size(); + break; + case json::value_t::number_integer: + st.num += static_cast(v.get()); + break; + case json::value_t::number_unsigned: + st.num += static_cast(v.get()); + break; + case json::value_t::number_float: + st.num += v.get(); + break; + default: + break; + } +} + +static void walk(const json& j, stats& st) +{ + ++st.nodes; + switch (j.type()) + { + case json::value_t::object: + for (const auto& kv : j.get_ref()) + { + st.str += kv.first.size(); + walk(kv.second, st); + } + break; + case json::value_t::array: + for (const auto& e : j.get_ref()) + { + walk(e, st); + } + break; + case json::value_t::string: + st.str += j.get_ref().size(); + break; + case json::value_t::number_integer: + st.num += static_cast(*j.get_ptr()); + break; + case json::value_t::number_unsigned: + st.num += static_cast(*j.get_ptr()); + break; + case json::value_t::number_float: + st.num += *j.get_ptr(); + break; + default: + break; + } +} + +static void walk(yyjson_val* v, stats& st) +{ + ++st.nodes; + switch (yyjson_get_type(v)) + { + case YYJSON_TYPE_OBJ: + { + std::size_t idx, max; + yyjson_val* k, * val; + yyjson_obj_foreach(v, idx, max, k, val) + { + st.str += yyjson_get_len(k); + walk(val, st); + } + break; + } + case YYJSON_TYPE_ARR: + { + std::size_t idx, max; + yyjson_val* val; + yyjson_arr_foreach(v, idx, max, val) + { + walk(val, st); + } + break; + } + case YYJSON_TYPE_STR: + st.str += yyjson_get_len(v); + break; + case YYJSON_TYPE_NUM: + if (yyjson_is_sint(v)) + { + st.num += static_cast(yyjson_get_sint(v)); + } + else if (yyjson_is_uint(v)) + { + st.num += static_cast(yyjson_get_uint(v)); + } + else + { + st.num += yyjson_get_real(v); + } + break; + default: + break; + } +} + +static void walk(simdjson::dom::element e, stats& st) +{ + ++st.nodes; + switch (e.type()) + { + case simdjson::dom::element_type::OBJECT: + for (auto f : simdjson::dom::object(e)) + { + st.str += f.key.size(); + walk(f.value, st); + } + break; + case simdjson::dom::element_type::ARRAY: + for (auto c : simdjson::dom::array(e)) + { + walk(c, st); + } + break; + case simdjson::dom::element_type::STRING: + st.str += std::string_view(e).size(); + break; + case simdjson::dom::element_type::INT64: + st.num += static_cast(int64_t(e)); + break; + case simdjson::dom::element_type::UINT64: + st.num += static_cast(uint64_t(e)); + break; + case simdjson::dom::element_type::DOUBLE: + st.num += double(e); + break; + default: + break; + } +} + +static void walk_od(simdjson::ondemand::value v, stats& st) +{ + ++st.nodes; + switch (v.type()) + { + case simdjson::ondemand::json_type::object: + for (auto f : v.get_object()) + { + st.str += std::string_view(f.unescaped_key()).size(); + walk_od(f.value(), st); + } + break; + case simdjson::ondemand::json_type::array: + for (auto c : v.get_array()) + { + walk_od(c.value(), st); + } + break; + case simdjson::ondemand::json_type::string: + st.str += std::string_view(v.get_string()).size(); + break; + case simdjson::ondemand::json_type::number: + { + simdjson::ondemand::number n = v.get_number(); + switch (n.get_number_type()) + { + case simdjson::ondemand::number_type::signed_integer: + st.num += static_cast(n.get_int64()); + break; + case simdjson::ondemand::number_type::unsigned_integer: + st.num += static_cast(n.get_uint64()); + break; + default: + st.num += n.get_double(); + break; + } + break; + } + case simdjson::ondemand::json_type::boolean: + (void)bool(v.get_bool()); + break; + default: + (void)v.is_null(); + break; + } +} + +#if JSON_VIEW_BENCH_BOOST +static void walk(const boost::json::value& v, stats& st) +{ + ++st.nodes; + switch (v.kind()) + { + case boost::json::kind::object: + for (const auto& kv : v.get_object()) + { + st.str += kv.key().size(); + walk(kv.value(), st); + } + break; + case boost::json::kind::array: + for (const auto& c : v.get_array()) + { + walk(c, st); + } + break; + case boost::json::kind::string: + st.str += v.get_string().size(); + break; + case boost::json::kind::int64: + st.num += static_cast(v.get_int64()); + break; + case boost::json::kind::uint64: + st.num += static_cast(v.get_uint64()); + break; + case boost::json::kind::double_: + st.num += v.get_double(); + break; + default: + break; + } +} +#endif + +// ---------------- selective access ---------------- +// twitter: per status id, user.screen_name, retweet_count +// citm: per performance id, eventId, #seatCategories; #events +// canada: type, features[0].geometry.type, #coordinates +// jeopardy: per question: round == "Final Jeopardy!", len(category) +// status (one tweet): id, user.screen_name, retweet_count +// rpc: method, params.minuend, id + +static double pick(const std::string& name, json_view r) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](json_view s) + { + acc += static_cast(s["id"].get()); + acc += static_cast(s["user"]["screen_name"].get_string().size()); + acc += static_cast(s["retweet_count"].get()); + }; + if (name == "twitter") + { + for (const json_view s : r["statuses"]) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + for (const json_view p : r["performances"]) + { + acc += static_cast(p["id"].get() + p["eventId"].get() + p["seatCategories"].size()); + } + acc += static_cast(r["events"].size()); + } + else if (name == "canada") + { + const json_view g = r["features"][0]["geometry"]; + acc += static_cast(r["type"].get_string().size() + g["type"].get_string().size() + g["coordinates"].size()); + } + else if (name == "jeopardy") + { + for (const json_view q : r) + { + acc += q["round"].get_string() == "Final Jeopardy!" ? 1 : 0; + acc += static_cast(q["category"].get_string().size()); + } + } + else if (name == "rpc") + { + acc += static_cast(r["method"].get_string().size()); + acc += static_cast(r["params"]["minuend"].get() + r["id"].get()); + } + return acc; +} + +static double pick(const std::string& name, const json& r) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](const json & s) + { + acc += static_cast(s["id"].get()); + acc += static_cast(s["user"]["screen_name"].get_ref().size()); + acc += static_cast(s["retweet_count"].get()); + }; + if (name == "twitter") + { + for (const auto& s : r["statuses"]) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + for (const auto& p : r["performances"]) + { + acc += static_cast(p["id"].get() + p["eventId"].get() + p["seatCategories"].size()); + } + acc += static_cast(r["events"].size()); + } + else if (name == "canada") + { + const json& g = r["features"][0]["geometry"]; + acc += static_cast(r["type"].get_ref().size() + g["type"].get_ref().size() + g["coordinates"].size()); + } + else if (name == "jeopardy") + { + for (const auto& q : r) + { + acc += q["round"].get_ref() == "Final Jeopardy!" ? 1 : 0; + acc += static_cast(q["category"].get_ref().size()); + } + } + else if (name == "rpc") + { + acc += static_cast(r["method"].get_ref().size()); + acc += static_cast(r["params"]["minuend"].get() + r["id"].get()); + } + return acc; +} + +static double pick(const std::string& name, yyjson_val* r) +{ + double acc = 0; + auto get = [](yyjson_val * o, const char* k) + { + return yyjson_obj_get(o, k); + }; + if (name == "twitter" || name == "status") + { + auto one = [&](yyjson_val * s) + { + acc += static_cast(yyjson_get_uint(get(s, "id"))); + acc += static_cast(yyjson_get_len(get(get(s, "user"), "screen_name"))); + acc += static_cast(yyjson_get_sint(get(s, "retweet_count"))); + }; + if (name == "twitter") + { + std::size_t idx, max; + yyjson_val* s; + yyjson_arr_foreach(get(r, "statuses"), idx, max, s) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + std::size_t idx, max; + yyjson_val* p; + yyjson_arr_foreach(get(r, "performances"), idx, max, p) + { + acc += static_cast(yyjson_get_uint(get(p, "id")) + yyjson_get_uint(get(p, "eventId")) + yyjson_arr_size(get(p, "seatCategories"))); + } + acc += static_cast(yyjson_obj_size(get(r, "events"))); + } + else if (name == "canada") + { + yyjson_val* g = get(yyjson_arr_get(get(r, "features"), 0), "geometry"); + acc += static_cast(yyjson_get_len(get(r, "type")) + yyjson_get_len(get(g, "type")) + yyjson_arr_size(get(g, "coordinates"))); + } + else if (name == "jeopardy") + { + std::size_t idx, max; + yyjson_val* q; + yyjson_arr_foreach(r, idx, max, q) + { + acc += yyjson_equals_str(get(q, "round"), "Final Jeopardy!") ? 1 : 0; + acc += static_cast(yyjson_get_len(get(q, "category"))); + } + } + else if (name == "rpc") + { + acc += static_cast(yyjson_get_len(get(r, "method"))); + acc += static_cast(yyjson_get_sint(get(get(r, "params"), "minuend")) + yyjson_get_sint(get(r, "id"))); + } + return acc; +} + +static double pick(const std::string& name, simdjson::dom::element r) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](simdjson::dom::element s) + { + acc += static_cast(uint64_t(s["id"])); + acc += static_cast(std::string_view(s["user"]["screen_name"]).size()); + acc += static_cast(int64_t(s["retweet_count"])); + }; + if (name == "twitter") + { + for (auto s : simdjson::dom::array(r["statuses"])) + { + one(s); + } + } + else + { + one(r); + } + } + else if (name == "citm_catalog") + { + for (auto p : simdjson::dom::array(r["performances"])) + { + acc += static_cast(uint64_t(p["id"]) + uint64_t(p["eventId"]) + simdjson::dom::array(p["seatCategories"]).size()); + } + acc += static_cast(simdjson::dom::object(r["events"]).size()); + } + else if (name == "canada") + { + auto g = r["features"].at(0)["geometry"]; + acc += static_cast(std::string_view(r["type"]).size() + std::string_view(g["type"]).size() + simdjson::dom::array(g["coordinates"]).size()); + } + else if (name == "jeopardy") + { + for (auto q : simdjson::dom::array(r)) + { + acc += std::string_view(q["round"]) == "Final Jeopardy!" ? 1 : 0; + acc += static_cast(std::string_view(q["category"]).size()); + } + } + else if (name == "rpc") + { + acc += static_cast(std::string_view(r["method"]).size()); + acc += static_cast(int64_t(r["params"]["minuend"]) + int64_t(r["id"])); + } + return acc; +} + +static double pick_od(const std::string& name, simdjson::ondemand::document& d) +{ + double acc = 0; + if (name == "twitter" || name == "status") + { + auto one = [&](simdjson::ondemand::object s) + { + acc += static_cast(uint64_t(s["id"])); + acc += static_cast(std::string_view(s["user"]["screen_name"]).size()); + acc += static_cast(int64_t(s["retweet_count"])); + }; + if (name == "twitter") + { + for (auto s : d["statuses"]) + { + one(s.get_object()); + } + } + else + { + one(d.get_object()); + } + } + else if (name == "citm_catalog") + { + simdjson::ondemand::object ev = d["events"].get_object(); + acc += static_cast(ev.count_fields()); + for (auto p : d["performances"]) + { + simdjson::ondemand::object o = p.get_object(); + const auto a = uint64_t(o["eventId"]) + uint64_t(o["id"]); + simdjson::ondemand::array sc = o["seatCategories"].get_array(); + acc += static_cast(a + sc.count_elements()); + } + } + else if (name == "canada") + { + acc += static_cast(std::string_view(d["type"]).size()); + auto g = d["features"].at(0)["geometry"]; + acc += static_cast(std::string_view(g["type"]).size()); + simdjson::ondemand::array co = g["coordinates"].get_array(); + acc += static_cast(co.count_elements()); + } + else if (name == "jeopardy") + { + for (auto q : d) + { + simdjson::ondemand::object o = q.get_object(); + acc += static_cast(std::string_view(o["category"]).size()); + acc += std::string_view(o["round"]) == "Final Jeopardy!" ? 1 : 0; + } + } + else if (name == "rpc") + { + acc += static_cast(std::string_view(d["method"]).size()); + acc += static_cast(int64_t(d["params"]["minuend"])); + acc += static_cast(int64_t(d["id"])); + } + return acc; +} + +// ---------------- harness ---------------- + +static std::string slurp(const std::string& p) +{ + std::ifstream f(p, std::ios::binary); + std::stringstream ss; + ss << f.rdbuf(); + return ss.str(); +} + +struct engine +{ + std::string name; + std::function fn; +}; + +int main(int argc, char** argv) +{ + if (argc < 2) + { + std::fprintf(stderr, "usage: %s [rounds] [document]\n", argv[0]); + return 1; + } + const std::string T = std::string(argv[1]) + "/"; + const int rounds = argc > 2 ? std::atoi(argv[2]) : 30; + const std::string only = argc > 3 ? argv[3] : ""; + struct doc + { + std::string name, text; + int batch; + }; + std::vector docs; + for (const char* f : + {"nativejson-benchmark/twitter.json", "nativejson-benchmark/citm_catalog.json", "nativejson-benchmark/canada.json", "jeopardy/jeopardy.json" + }) + { + std::string n = std::string(f).substr(std::string(f).find('/') + 1); + docs.push_back({n.substr(0, n.size() - 5), slurp(T + f), 1}); + } + docs.push_back({"status", json::parse(docs[0].text)["statuses"][0].dump(), 200}); + docs.push_back({"rpc", R"({"jsonrpc": "2.0", "method": "subtract", "params": {"minuend": 42, "subtrahend": 23}, "id": 3})", 5000}); + + // correctness cross-check of the workloads + for (const auto& dc : docs) + { + stats a, b, c, dd; + walk(json::parse(dc.text), a); + auto d = json_document::parse(dc.text); + walk(d.root(), b); + yyjson_doc* y = yyjson_read(dc.text.data(), dc.text.size(), 0); + walk(yyjson_doc_get_root(y), c); + simdjson::dom::parser p; + walk(p.parse(dc.text).value(), dd); + const bool ok = a.nodes == b.nodes && a.nodes == c.nodes && a.nodes == dd.nodes && a.str == b.str && a.str == c.str && a.str == dd.str + && std::fabs(a.num - b.num) <= 1e-9 * std::fabs(a.num) && std::fabs(a.num - c.num) <= 1e-9 * std::fabs(a.num); + const double pa = pick(dc.name, json::parse(dc.text)), pb = pick(dc.name, d.root()), pc = pick(dc.name, yyjson_doc_get_root(y)), pd = pick(dc.name, p.parse(dc.text).value()); + std::printf("check %-13s traverse %s select %s\n", dc.name.c_str(), ok ? "OK" : "MISMATCH", (pa == pb && pa == pc && pa == pd) ? "OK" : "MISMATCH"); + yyjson_doc_free(y); + } + + std::FILE* csv = std::fopen("bench_view.csv", "w"); + std::fprintf(csv, "doc,bytes,workload,engine,ns\n"); + json_document reused; + simdjson::dom::parser sj; + simdjson::ondemand::parser od; + for (const auto& dc : docs) + { + if (!only.empty() && dc.name != only) + { + continue; + } + const std::string& s = dc.text; + const simdjson::padded_string ps(s); + const std::string name = dc.name; + std::vector>> workloads; + + workloads.push_back({"parse", { + {"json_view", [&] { auto d = json_document::parse(s); g_sink = static_cast(d.node_count()); }}, + {"json_view (reused)", [&] { reused.read(s); g_sink = static_cast(reused.node_count()); }}, + {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = static_cast(yyjson_doc_get_val_count(d)); yyjson_doc_free(d); }}, + {"simdjson DOM", [&] { auto e = sj.parse(ps).value_unsafe(); g_sink = e.is_object(); }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); g_sink = v.is_object(); }}, +#endif + {"json::parse", [&] { json j = json::parse(s); g_sink = static_cast(j.size()); }}, + }}); + workloads.push_back({"traverse", { + {"json_view", [&] { auto d = json_document::parse(s); stats st; walk(d.root(), st); g_sink = st.num; }}, + {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); stats st; walk(yyjson_doc_get_root(d), st); g_sink = st.num; yyjson_doc_free(d); }}, + {"simdjson DOM", [&] { stats st; walk(sj.parse(ps).value_unsafe(), st); g_sink = st.num; }}, + {"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); stats st; walk_od(d.get_value().value_unsafe(), st); g_sink = st.num; }}, +#if JSON_VIEW_BENCH_BOOST + {"Boost.JSON", [&] { boost::json::monotonic_resource mr; auto v = boost::json::parse(s, &mr); stats st; walk(v, st); g_sink = st.num; }}, +#endif + {"json::parse", [&] { json j = json::parse(s); stats st; walk(j, st); g_sink = st.num; }}, + }}); + workloads.push_back({"select", { + {"json_view", [&] { auto d = json_document::parse(s); g_sink = pick(name, d.root()); }}, + {"yyjson", [&] { yyjson_doc* d = yyjson_read(s.data(), s.size(), 0); g_sink = pick(name, yyjson_doc_get_root(d)); yyjson_doc_free(d); }}, + {"simdjson DOM", [&] { g_sink = pick(name, sj.parse(ps).value_unsafe()); }}, + {"simdjson OD", [&] { auto d = od.iterate(ps).value_unsafe(); g_sink = pick_od(name, d); }}, + {"json::parse", [&] { json j = json::parse(s); g_sink = pick(name, j); }}, + }}); + { + // serialization of an already parsed document + static json_document vd; + vd.read(s); + static yyjson_doc* yd = nullptr; + if (yd) + { + yyjson_doc_free(yd); + } + yd = yyjson_read(s.data(), s.size(), 0); + static simdjson::dom::parser sjd; + static simdjson::dom::element se; + se = sjd.parse(ps).value_unsafe(); + static json jd; + jd = json::parse(s); + workloads.push_back({"dump", { + {"json_view", [&] { std::string o = vd.root().dump(); g_sink = static_cast(o.size()); }}, + {"yyjson", [&] { std::size_t n = 0; char* o = yyjson_write(yd, 0, &n); g_sink = static_cast(n); std::free(o); }}, + {"simdjson DOM", [&] { std::string o = simdjson::to_string(se); g_sink = static_cast(o.size()); }}, + {"json::parse", [&] { std::string o = jd.dump(); g_sink = static_cast(o.size()); }}, + }}); + } + + for (auto& wl : workloads) + { + std::vector best(wl.second.size(), 1e300); + const int r = s.size() > 10000000 ? std::max(3, rounds / 5) : rounds; + for (int i = 0; i < r; ++i) + { + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + const auto t0 = std::chrono::steady_clock::now(); + for (int b = 0; b < dc.batch; ++b) + { + wl.second[k].fn(); + } + const double ns = std::chrono::duration(std::chrono::steady_clock::now() - t0).count() / dc.batch; + best[k] = std::min(best[k], ns); + } + } + const double ref = best[0]; + std::printf("%-13s %-9s", dc.name.c_str(), wl.first.c_str()); + for (std::size_t k = 0; k < wl.second.size(); ++k) + { + const double us = best[k] / 1e3; + std::printf(" %s %s%s (%.2fx)", wl.second[k].name.c_str(), us >= 100 ? "" : "", (us >= 1000 ? std::to_string(static_cast(us)) + "us" : (std::to_string(us).substr(0, 5) + "us")).c_str(), best[k] / ref); + std::fprintf(csv, "%s,%zu,%s,%s,%.1f\n", dc.name.c_str(), s.size(), wl.first.c_str(), wl.second[k].name.c_str(), best[k]); + } + std::printf("\n"); + std::fflush(stdout); + } + } + std::fclose(csv); +} diff --git a/tests/benchmarks/json_view/compare.py b/tests/benchmarks/json_view/compare.py new file mode 100755 index 000000000..e28d419c6 --- /dev/null +++ b/tests/benchmarks/json_view/compare.py @@ -0,0 +1,295 @@ +#!/usr/bin/env python3 +# __ _____ _____ _____ +# __| | __| | | | JSON for Modern C++ (supporting code) +# | | |__ | | | | | | version 3.12.0 +# |_____|_____|_____|_|___| https://github.com/nlohmann/json +# +# SPDX-FileCopyrightText: 2013-2026 Niels Lohmann +# SPDX-License-Identifier: MIT + +"""Compare json_view with yyjson, simdjson, Boost.JSON, and json::parse. + +Builds bench_view.cpp and bench_corpus.cpp against the include/ directory of +this checkout, runs them, and writes the results with everything needed to +reproduce them (date, commit, CPU, OS, compiler, library versions, flags) to +results/-.md and .csv next to this script. + +The other libraries come from the system (--system, the default: pkg-config +or Homebrew) or are downloaded as pinned releases and checked against their +SHA-256 (--download). Boost.JSON is optional: without Boost headers, its +columns are skipped, and the results say so. + +Only the Python 3 standard library is used; a C++17 compiler is needed. +""" + +import argparse +import datetime +import hashlib +import os +import platform +import re +import shlex +import shutil +import subprocess +import sys +import tarfile +import urllib.request + +HERE = os.path.dirname(os.path.abspath(__file__)) +REPO = os.path.abspath(os.path.join(HERE, '..', '..', '..')) + +# pinned releases for --download; the hashes are those of the archives +PINNED = { + 'yyjson': { + 'version': '0.13.0', + 'url': 'https://github.com/ibireme/yyjson/archive/refs/tags/0.13.0.tar.gz', + 'sha256': None, # TODO: pin before the first published comparison + }, + 'simdjson': { + 'version': '4.6.11', + 'url': 'https://github.com/simdjson/simdjson/archive/refs/tags/v4.6.11.tar.gz', + 'sha256': None, # TODO: pin before the first published comparison + }, + 'boost': { + 'version': '1.92.0', + 'url': 'https://archives.boost.io/release/1.92.0/source/boost_1_92_0.tar.gz', + 'sha256': None, # TODO: pin before the first published comparison + }, +} + +# the documents of bench_view.cpp, relative to the json_test_data directory +DEFAULT_CORPUS = [ + 'nativejson-benchmark/twitter.json', + 'nativejson-benchmark/citm_catalog.json', + 'nativejson-benchmark/canada.json', + 'jeopardy/jeopardy.json', +] + + +def run(cmd, **kwargs): + print('+ ' + ' '.join(shlex.quote(c) for c in cmd), flush=True) + return subprocess.run(cmd, check=True, **kwargs) + + +def output(cmd): + try: + return subprocess.run(cmd, check=True, capture_output=True, text=True).stdout.strip() + except (OSError, subprocess.CalledProcessError): + return '' + + +# --------------------------------------------------------------------------- +# libraries +# --------------------------------------------------------------------------- + +class Library: + """include directories, sources to compile, and linker flags of a library""" + + def __init__(self, name, include=None, sources=None, link=None, version=''): + self.name = name + self.include = include or [] + self.sources = sources or [] + self.link = link or [] + self.version = version + + +def header_version(path, pattern): + try: + with open(path, encoding='utf-8', errors='replace') as f: + m = re.search(pattern, f.read()) + return m.group(1) if m else '' + except OSError: + return '' + + +def library_version(name, include_dirs): + patterns = { + 'yyjson': ('yyjson.h', r'#define\s+YYJSON_VERSION_STRING\s+"([^"]+)"'), + 'simdjson': ('simdjson.h', r'#define\s+SIMDJSON_VERSION\s+"?([0-9.]+)"?'), + 'boost': (os.path.join('boost', 'version.hpp'), r'#define\s+BOOST_LIB_VERSION\s+"([^"]+)"'), + } + header, pattern = patterns[name] + for d in include_dirs: + v = header_version(os.path.join(d, header), pattern) + if v: + return v.replace('_', '.') + return '' + + +def system_library(name): + """a library found with pkg-config or Homebrew, or None""" + flags = output(['pkg-config', '--cflags', '--libs', name]).split() + if flags: + include = [f[2:] for f in flags if f.startswith('-I')] + link = [f for f in flags if f.startswith('-L') or f.startswith('-l')] + libdirs = [f[2:] for f in link if f.startswith('-L')] + link += ['-Wl,-rpath,' + d for d in libdirs] + return Library(name, include, [], link, library_version(name, include)) + prefix = output(['brew', '--prefix', name]) if shutil.which('brew') else '' + if prefix and os.path.isdir(os.path.join(prefix, 'include')): + include = [os.path.join(prefix, 'include')] + link = [] + if name != 'boost': + lib = os.path.join(prefix, 'lib') + link = ['-L' + lib, '-l' + name, '-Wl,-rpath,' + lib] + return Library(name, include, [], link, library_version(name, include)) + if name == 'boost': + for d in ['/usr/include', '/usr/local/include']: + if os.path.isfile(os.path.join(d, 'boost', 'json.hpp')): + return Library(name, [d], [], [], library_version(name, [d])) + return None + + +def download_library(name, work): + """a pinned release, downloaded and checked, or an error""" + pin = PINNED[name] + if not pin['sha256']: + sys.exit(f'error: no SHA-256 pinned for {name} {pin["version"]} yet; use --system') + archive = os.path.join(work, 'download', os.path.basename(pin['url'])) + os.makedirs(os.path.dirname(archive), exist_ok=True) + if not os.path.isfile(archive): + print(f'downloading {pin["url"]}', flush=True) + urllib.request.urlretrieve(pin['url'], archive) + with open(archive, 'rb') as f: + digest = hashlib.sha256(f.read()).hexdigest() + if digest != pin['sha256']: + sys.exit(f'error: SHA-256 of {archive} is {digest}, expected {pin["sha256"]}') + target = os.path.join(work, 'download', f'{name}-{pin["version"]}') + if not os.path.isdir(target): + with tarfile.open(archive) as t: + t.extractall(os.path.join(work, 'download')) # noqa: S202 (checked archive) + src = [os.path.join(work, 'download', d) for d in os.listdir(os.path.join(work, 'download')) + if d.lower().startswith(name) and os.path.isdir(os.path.join(work, 'download', d))][0] + if name == 'yyjson': + return Library(name, [os.path.join(src, 'src')], [os.path.join(src, 'src', 'yyjson.c')], [], pin['version']) + if name == 'simdjson': + single = os.path.join(src, 'singleheader') + return Library(name, [single], [os.path.join(single, 'simdjson.cpp')], [], pin['version']) + return Library(name, [src], [], [], pin['version']) + + +# --------------------------------------------------------------------------- +# machine description +# --------------------------------------------------------------------------- + +def cpu_model(): + if sys.platform == 'darwin': + return output(['sysctl', '-n', 'machdep.cpu.brand_string']) + try: + with open('/proc/cpuinfo', encoding='utf-8') as f: + for line in f: + if line.startswith('model name') or line.startswith('Model'): + return line.split(':', 1)[1].strip() + except OSError: + pass + return platform.processor() + + +def git_commit(): + commit = output(['git', '-C', REPO, 'rev-parse', '--short=12', 'HEAD']) + dirty = output(['git', '-C', REPO, 'status', '--porcelain', '--untracked-files=no']) + return commit + (' (with local changes)' if dirty else '') + + +# --------------------------------------------------------------------------- +# main +# --------------------------------------------------------------------------- + +def main(): + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument('--data', required=True, help='json_test_data directory (with nativejson-benchmark/ and jeopardy/)') + ap.add_argument('--download', action='store_true', help='use pinned downloads instead of system libraries') + ap.add_argument('--no-boost', action='store_true', help='skip Boost.JSON') + ap.add_argument('--native', action='store_true', help='compile for this CPU (-march=native / -mcpu=native)') + ap.add_argument('--rounds', type=int, default=30, help='rounds of bench_view (default: 30)') + ap.add_argument('--corpus', nargs='*', default=[], help='more files for bench_corpus') + ap.add_argument('--build-dir', default=os.path.join(HERE, 'build'), help='where to build (default: build/ next to this script)') + args = ap.parse_args() + + cxx = os.environ.get('CXX', 'c++') + cc = os.environ.get('CC', 'cc') + os.makedirs(args.build_dir, exist_ok=True) + + libs = {} + for name in ['yyjson', 'simdjson', 'boost']: + if name == 'boost' and args.no_boost: + continue + lib = download_library(name, args.build_dir) if args.download else system_library(name) + if lib is None and name != 'boost': + sys.exit(f'error: {name} not found; install it, or use --download') + if lib is not None: + libs[name] = lib + with_boost = 'boost' in libs + if not with_boost: + print('Boost.JSON not found: its columns are skipped', flush=True) + + flags = ['-std=c++17', '-O3', '-DNDEBUG', f'-DJSON_VIEW_BENCH_BOOST={1 if with_boost else 0}'] + if args.native: + flags.append('-mcpu=native' if platform.machine().lower() in ('arm64', 'aarch64') else '-march=native') + include = ['-I' + os.path.join(REPO, 'include')] + ['-I' + d for lib in libs.values() for d in lib.include] + link = [f for lib in libs.values() for f in lib.link] + + # C sources of downloaded libraries are compiled once + objects = [] + for lib in libs.values(): + for src in lib.sources: + obj = os.path.join(args.build_dir, os.path.basename(src) + '.o') + compiler = cc if src.endswith('.c') else cxx + run([compiler] + (['-std=c++17'] if compiler == cxx else []) + ['-O3', '-DNDEBUG', '-c', src, '-o', obj] + + ['-I' + d for d in lib.include]) + objects.append(obj) + + binaries = {} + for bench in ['bench_view', 'bench_corpus']: + exe = os.path.join(args.build_dir, bench) + run([cxx] + flags + include + [os.path.join(HERE, bench + '.cpp')] + objects + link + ['-o', exe]) + binaries[bench] = exe + + # run: bench_view on its documents, bench_corpus on those and the given files + corpus = [os.path.join(args.data, f) for f in DEFAULT_CORPUS] + args.corpus + outputs = {} + outputs['bench_view'] = run([binaries['bench_view'], args.data, str(args.rounds)], cwd=args.build_dir, + capture_output=True, text=True).stdout + outputs['bench_corpus'] = run([binaries['bench_corpus']] + corpus, cwd=args.build_dir, + capture_output=True, text=True).stdout + for name, text in outputs.items(): + print(text) + + # results with their metadata + now = datetime.datetime.now() + host = re.sub(r'[^A-Za-z0-9-]+', '-', platform.node().split('.')[0]) or 'host' + stem = os.path.join(HERE, 'results', f'{now:%Y-%m-%d}-{host}') + os.makedirs(os.path.dirname(stem), exist_ok=True) + meta = [ + ('date', f'{now:%Y-%m-%d %H:%M}'), + ('commit', git_commit()), + ('CPU', cpu_model()), + ('OS', f'{platform.system()} {platform.release()} ({platform.machine()})'), + ('compiler', output([cxx, '--version']).splitlines()[0] if output([cxx, '--version']) else cxx), + ('flags', ' '.join(flags)), + ('yyjson', libs['yyjson'].version), + ('simdjson', libs['simdjson'].version), + ('Boost.JSON', libs['boost'].version if with_boost else 'skipped (not found)'), + ('libraries from', 'pinned downloads' if args.download else 'the system'), + ('rounds', str(args.rounds)), + ] + with open(stem + '.md', 'w', encoding='utf-8') as f: + f.write(f'# json_view comparison, {now:%Y-%m-%d}\n\n') + f.write('Generated by `tests/benchmarks/json_view/compare.py`; best of the interleaved rounds.\n\n') + f.write('| | |\n|---|---|\n') + for key, value in meta: + f.write(f'| {key} | {value} |\n') + for name, text in outputs.items(): + f.write(f'\n## {name}\n\n```\n{text.rstrip()}\n```\n') + with open(stem + '.csv', 'w', encoding='utf-8') as out: + out.write(''.join(f'# {key}: {value}\n' for key, value in meta)) + for name in ['bench_view', 'bench_corpus']: + path = os.path.join(args.build_dir, name + '.csv') + if os.path.isfile(path): + with open(path, encoding='utf-8') as f: + out.write(f'# {name}\n' + f.read()) + print(f'results: {stem}.md, {stem}.csv') + + +if __name__ == '__main__': + main()