| // __ _____ _____ _____ |
| // __| | __| | | | JSON for Modern C++ (supporting code) |
| // | | |__ | | | | | | version 3.12.0 |
| // |_____|_____|_____|_|___| https://github.com/nlohmann/json |
| // |
| // SPDX-FileCopyrightText: 2013-2026 Niels Lohmann <https://nlohmann.me> |
| // SPDX-License-Identifier: MIT |
| |
| #include "doctest_compatibility.h" |
| |
| // capture whether JSON_DELETE_DEPRECATED_FUNCTIONS was enabled on the command |
| // line *before* including json.hpp, since the library #undefs it once the header |
| // has been fully processed (see include/nlohmann/detail/macro_unscope.hpp); the |
| // tests of deprecated functions are skipped if these functions are deleted |
| #if defined(JSON_DELETE_DEPRECATED_FUNCTIONS) && (JSON_DELETE_DEPRECATED_FUNCTIONS == 1) |
| #define JSON_TEST_DEPRECATED_FUNCTIONS_DELETED |
| #endif |
| |
| #include <nlohmann/json.hpp> |
| using nlohmann::json; |
| |
| #include <array> |
| #include <sstream> |
| #include <iomanip> |
| |
| #include "test_utils.hpp" |
| |
| TEST_CASE("serialization") |
| { |
| SECTION("operator<<") |
| { |
| SECTION("no given width") |
| { |
| std::stringstream ss; |
| const json j = {"foo", 1, 2, 3, false, {{"one", 1}}}; |
| ss << j; |
| CHECK(ss.str() == "[\"foo\",1,2,3,false,{\"one\":1}]"); |
| } |
| |
| SECTION("given width") |
| { |
| std::stringstream ss; |
| const json j = {"foo", 1, 2, 3, false, {{"one", 1}}}; |
| ss << std::setw(4) << j; |
| CHECK(ss.str() == |
| "[\n \"foo\",\n 1,\n 2,\n 3,\n false,\n {\n \"one\": 1\n }\n]"); |
| } |
| |
| SECTION("given fill") |
| { |
| std::stringstream ss; |
| const json j = {"foo", 1, 2, 3, false, {{"one", 1}}}; |
| ss << std::setw(1) << std::setfill('\t') << j; |
| CHECK(ss.str() == |
| "[\n\t\"foo\",\n\t1,\n\t2,\n\t3,\n\tfalse,\n\t{\n\t\t\"one\": 1\n\t}\n]"); |
| } |
| } |
| |
| #ifndef JSON_TEST_DEPRECATED_FUNCTIONS_DELETED |
| SECTION("operator>>") |
| { |
| SECTION("no given width") |
| { |
| std::stringstream ss; |
| const json j = {"foo", 1, 2, 3, false, {{"one", 1}}}; |
| j >> ss; |
| CHECK(ss.str() == "[\"foo\",1,2,3,false,{\"one\":1}]"); |
| } |
| |
| SECTION("given width") |
| { |
| std::stringstream ss; |
| const json j = {"foo", 1, 2, 3, false, {{"one", 1}}}; |
| ss.width(4); |
| j >> ss; |
| CHECK(ss.str() == |
| "[\n \"foo\",\n 1,\n 2,\n 3,\n false,\n {\n \"one\": 1\n }\n]"); |
| } |
| |
| SECTION("given fill") |
| { |
| std::stringstream ss; |
| const json j = {"foo", 1, 2, 3, false, {{"one", 1}}}; |
| ss.width(1); |
| ss.fill('\t'); |
| j >> ss; |
| CHECK(ss.str() == |
| "[\n\t\"foo\",\n\t1,\n\t2,\n\t3,\n\tfalse,\n\t{\n\t\t\"one\": 1\n\t}\n]"); |
| } |
| } |
| #endif |
| |
| SECTION("dump") |
| { |
| SECTION("invalid character") |
| { |
| const json j = "ä\xA9ü"; |
| |
| // dump() is nodiscard; the exception is thrown by dump() itself before it would return |
| CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&); |
| CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"äü\""); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"ä\xEF\xBF\xBDü\""); |
| CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"\\u00e4\\ufffd\\u00fc\""); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"ä\xA9ü\""); |
| CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"\\u00e4\xA9\\u00fc\""); |
| } |
| |
| SECTION("invalid character (regression guard for shared UTF-8 decoder, see #5529)") |
| { |
| // dump_escaped_impl() now calls the UTF-8 decoder shared with the |
| // binary readers (detail::decode() in string_utils.hpp) instead |
| // of a private copy; the exact type_error.316 message/behavior |
| // must stay byte-for-byte the same as before that extraction |
| const json j = "ä\xA9ü"; |
| CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 2: 0xA9", json::type_error&); |
| } |
| |
| SECTION("ending with incomplete character") |
| { |
| const json j = "123\xC2"; |
| |
| // dump() is nodiscard; the exception is thrown by dump() itself before it would return |
| CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] incomplete UTF-8 string; last byte: 0xC2", json::type_error&); |
| CHECK_THROWS_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), json::type_error&); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123\""); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\""); |
| CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd\""); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"123\xC2\""); |
| CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"123\xC2\""); |
| } |
| |
| SECTION("unexpected character") |
| { |
| const json j = "123\xF1\xB0\x34\x35\x36"; |
| |
| // dump() is nodiscard; the exception is thrown by dump() itself before it would return |
| CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0x34", json::type_error&); |
| CHECK_THROWS_AS(utils::ignore_return_value(j.dump(1, ' ', false, json::error_handler_t::strict)), json::type_error&); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"123456\""); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"123\xEF\xBF\xBD\x34\x35\x36\""); |
| CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"123\\ufffd456\""); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"123\xF1\xB0\x34\x35\x36\""); |
| CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"123\xF1\xB0\x34\x35\x36\""); |
| } |
| |
| SECTION("keep: valid characters are still escaped") |
| { |
| // an invalid byte followed by characters that must be escaped |
| const json j = "\xC2\"\\\n\xFF\x05"; |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xC2\\\"\\\\\\n\xFF\\u0005\""); |
| CHECK(j.dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xC2\\\"\\\\\\n\xFF\\u0005\""); |
| } |
| |
| SECTION("keep: truncated multibyte sequences") |
| { |
| CHECK(json("\xF0\x9F\x98").dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xF0\x9F\x98\""); |
| CHECK(json("\xF0\x9F\x98").dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xF0\x9F\x98\""); |
| CHECK(json("\xF0\x9F\x98" "a").dump(-1, ' ', false, json::error_handler_t::keep) == "\"\xF0\x9F\x98" "a\""); |
| CHECK(json("\xF0\x9F\x98" "a").dump(-1, ' ', true, json::error_handler_t::keep) == "\"\xF0\x9F\x98" "a\""); |
| } |
| |
| SECTION("keep: long string with many invalid bytes") |
| { |
| // exceeds the internal string buffer several times |
| std::string input; |
| std::string expected = "\""; |
| for (int i = 0; i < 2000; ++i) |
| { |
| input += "\xFF\xE2\x82\n\xC3\xA4"; |
| expected += "\xFF\xE2\x82\\n\xC3\xA4"; |
| } |
| expected += "\""; |
| const json j = input; |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::keep) == expected); |
| } |
| |
| SECTION("U+FFFD Substitution of Maximal Subparts") |
| { |
| // Some tests (mostly) from |
| // https://www.unicode.org/versions/Unicode11.0.0/ch03.pdf |
| // Section 3.9 -- U+FFFD Substitution of Maximal Subparts |
| |
| auto test = [&](std::string const & input, std::string const & expected) |
| { |
| const json j = input; |
| CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"" + expected + "\""); |
| }; |
| |
| test("\xC2", "\\ufffd"); |
| test("\xC2\x41\x42", "\\ufffd" "\x41" "\x42"); |
| test("\xC2\xF4", "\\ufffd" "\\ufffd"); |
| |
| test("\xF0\x80\x80\x41", "\\ufffd" "\\ufffd" "\\ufffd" "\x41"); |
| test("\xF1\x80\x80\x41", "\\ufffd" "\x41"); |
| test("\xF2\x80\x80\x41", "\\ufffd" "\x41"); |
| test("\xF3\x80\x80\x41", "\\ufffd" "\x41"); |
| test("\xF4\x80\x80\x41", "\\ufffd" "\x41"); |
| test("\xF5\x80\x80\x41", "\\ufffd" "\\ufffd" "\\ufffd" "\x41"); |
| |
| test("\xF0\x90\x80\x41", "\\ufffd" "\x41"); |
| test("\xF1\x90\x80\x41", "\\ufffd" "\x41"); |
| test("\xF2\x90\x80\x41", "\\ufffd" "\x41"); |
| test("\xF3\x90\x80\x41", "\\ufffd" "\x41"); |
| test("\xF4\x90\x80\x41", "\\ufffd" "\\ufffd" "\\ufffd" "\x41"); |
| test("\xF5\x90\x80\x41", "\\ufffd" "\\ufffd" "\\ufffd" "\x41"); |
| |
| test("\xC0\xAF\xE0\x80\xBF\xF0\x81\x82\x41", "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\x41"); |
| test("\xED\xA0\x80\xED\xBF\xBF\xED\xAF\x41", "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\x41"); |
| test("\xF4\x91\x92\x93\xFF\x41\x80\xBF\x42", "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\x41" "\\ufffd""\\ufffd" "\x42"); |
| test("\xE1\x80\xE2\xF0\x91\x92\xF1\xBF\x41", "\\ufffd" "\\ufffd" "\\ufffd" "\\ufffd" "\x41"); |
| } |
| } |
| |
| SECTION("to_string") |
| { |
| auto test = [&](std::string const & input, std::string const & expected) |
| { |
| using std::to_string; |
| const json j = input; |
| CHECK(to_string(j) == "\"" + expected + "\""); |
| }; |
| |
| test(R"({"x":5,"y":6})", R"({\"x\":5,\"y\":6})"); |
| test("{\"x\":[10,null,null,null]}", R"({\"x\":[10,null,null,null]})"); |
| test("test", "test"); |
| test("[3,\"false\",false]", R"([3,\"false\",false])"); |
| } |
| } |
| |
| TEST_CASE_TEMPLATE("serialization for extreme integer values", T, int32_t, uint32_t, int64_t, uint64_t) // NOLINT(readability-math-missing-parentheses, bugprone-throwing-static-initialization) |
| { |
| SECTION("minimum") |
| { |
| constexpr auto minimum = (std::numeric_limits<T>::min)(); |
| const json j = minimum; |
| CHECK(j.dump() == std::to_string(minimum)); |
| } |
| |
| SECTION("maximum") |
| { |
| constexpr auto maximum = (std::numeric_limits<T>::max)(); |
| const json j = maximum; |
| CHECK(j.dump() == std::to_string(maximum)); |
| } |
| } |
| |
| TEST_CASE("dump with binary values") |
| { |
| auto binary = json::binary({1, 2, 3, 4}); |
| auto binary_empty = json::binary({}); |
| auto binary_with_subtype = json::binary({1, 2, 3, 4}, 128); |
| auto binary_empty_with_subtype = json::binary({}, 128); |
| |
| const json object = {{"key", binary}}; |
| const json object_empty = {{"key", binary_empty}}; |
| const json object_with_subtype = {{"key", binary_with_subtype}}; |
| const json object_empty_with_subtype = {{"key", binary_empty_with_subtype}}; |
| |
| const json array = {"value", 1, binary}; |
| const json array_empty = {"value", 1, binary_empty}; |
| const json array_with_subtype = {"value", 1, binary_with_subtype}; |
| const json array_empty_with_subtype = {"value", 1, binary_empty_with_subtype}; |
| |
| SECTION("normal") |
| { |
| CHECK(binary.dump() == "{\"bytes\":[1,2,3,4],\"subtype\":null}"); |
| CHECK(binary_empty.dump() == "{\"bytes\":[],\"subtype\":null}"); |
| CHECK(binary_with_subtype.dump() == "{\"bytes\":[1,2,3,4],\"subtype\":128}"); |
| CHECK(binary_empty_with_subtype.dump() == "{\"bytes\":[],\"subtype\":128}"); |
| |
| CHECK(object.dump() == "{\"key\":{\"bytes\":[1,2,3,4],\"subtype\":null}}"); |
| CHECK(object_empty.dump() == "{\"key\":{\"bytes\":[],\"subtype\":null}}"); |
| CHECK(object_with_subtype.dump() == "{\"key\":{\"bytes\":[1,2,3,4],\"subtype\":128}}"); |
| CHECK(object_empty_with_subtype.dump() == "{\"key\":{\"bytes\":[],\"subtype\":128}}"); |
| |
| CHECK(array.dump() == "[\"value\",1,{\"bytes\":[1,2,3,4],\"subtype\":null}]"); |
| CHECK(array_empty.dump() == "[\"value\",1,{\"bytes\":[],\"subtype\":null}]"); |
| CHECK(array_with_subtype.dump() == "[\"value\",1,{\"bytes\":[1,2,3,4],\"subtype\":128}]"); |
| CHECK(array_empty_with_subtype.dump() == "[\"value\",1,{\"bytes\":[],\"subtype\":128}]"); |
| } |
| |
| SECTION("pretty-printed") |
| { |
| CHECK(binary.dump(4) == "{\n" |
| " \"bytes\": [1, 2, 3, 4],\n" |
| " \"subtype\": null\n" |
| "}"); |
| CHECK(binary_empty.dump(4) == "{\n" |
| " \"bytes\": [],\n" |
| " \"subtype\": null\n" |
| "}"); |
| CHECK(binary_with_subtype.dump(4) == "{\n" |
| " \"bytes\": [1, 2, 3, 4],\n" |
| " \"subtype\": 128\n" |
| "}"); |
| CHECK(binary_empty_with_subtype.dump(4) == "{\n" |
| " \"bytes\": [],\n" |
| " \"subtype\": 128\n" |
| "}"); |
| |
| CHECK(object.dump(4) == "{\n" |
| " \"key\": {\n" |
| " \"bytes\": [1, 2, 3, 4],\n" |
| " \"subtype\": null\n" |
| " }\n" |
| "}"); |
| CHECK(object_empty.dump(4) == "{\n" |
| " \"key\": {\n" |
| " \"bytes\": [],\n" |
| " \"subtype\": null\n" |
| " }\n" |
| "}"); |
| CHECK(object_with_subtype.dump(4) == "{\n" |
| " \"key\": {\n" |
| " \"bytes\": [1, 2, 3, 4],\n" |
| " \"subtype\": 128\n" |
| " }\n" |
| "}"); |
| CHECK(object_empty_with_subtype.dump(4) == "{\n" |
| " \"key\": {\n" |
| " \"bytes\": [],\n" |
| " \"subtype\": 128\n" |
| " }\n" |
| "}"); |
| |
| CHECK(array.dump(4) == "[\n" |
| " \"value\",\n" |
| " 1,\n" |
| " {\n" |
| " \"bytes\": [1, 2, 3, 4],\n" |
| " \"subtype\": null\n" |
| " }\n" |
| "]"); |
| CHECK(array_empty.dump(4) == "[\n" |
| " \"value\",\n" |
| " 1,\n" |
| " {\n" |
| " \"bytes\": [],\n" |
| " \"subtype\": null\n" |
| " }\n" |
| "]"); |
| CHECK(array_with_subtype.dump(4) == "[\n" |
| " \"value\",\n" |
| " 1,\n" |
| " {\n" |
| " \"bytes\": [1, 2, 3, 4],\n" |
| " \"subtype\": 128\n" |
| " }\n" |
| "]"); |
| CHECK(array_empty_with_subtype.dump(4) == "[\n" |
| " \"value\",\n" |
| " 1,\n" |
| " {\n" |
| " \"bytes\": [],\n" |
| " \"subtype\": 128\n" |
| " }\n" |
| "]"); |
| } |
| } |
| |
| TEST_CASE("dump for basic_json with long double number_float_t") |
| { |
| // Custom basic_json instantiation with long double as NumberFloatType. |
| // On platforms where long double is wider than double (e.g. GCC/Clang on |
| // Linux/macOS x86_64), dump() goes through the snprintf path in |
| // serializer::dump_float(x, std::false_type). That branch must use the |
| // "%.*Lg" format specifier; using "%.*g" with a long double argument is |
| // undefined behavior and corrupts the output. |
| using long_double_json = nlohmann::json::with_float_t<long double>; |
| |
| SECTION("round-trip dump/parse") |
| { |
| constexpr std::array<long double, 13> values = |
| { |
| { |
| 0.0L, -0.0L, 1.0L, -1.0L, |
| 0.5L, -0.5L, 1.5L, -2.25L, |
| 1.23e45L, 1.23e-45L, |
| (std::numeric_limits<long double>::min)(), |
| std::numeric_limits<long double>::lowest(), |
| (std::numeric_limits<long double>::max)() |
| } |
| }; |
| |
| for (long double v : values) |
| { |
| const long_double_json j = v; |
| const auto s = j.dump(); |
| const auto j2 = long_double_json::parse(s); |
| CHECK(j2.template get<long double>() == v); |
| } |
| } |
| |
| SECTION("exact dump string for simple values") |
| { |
| CHECK(long_double_json(0.5L).dump() == "0.5"); |
| CHECK(long_double_json(-0.5L).dump() == "-0.5"); |
| CHECK(long_double_json(1.5L).dump() == "1.5"); |
| CHECK(long_double_json(-2.25L).dump() == "-2.25"); |
| CHECK(long_double_json(0.0L).dump() == "0.0"); |
| CHECK(long_double_json(1.0L).dump() == "1.0"); |
| CHECK(long_double_json(-1.0L).dump() == "-1.0"); |
| CHECK(long_double_json(100.0L).dump() == "100.0"); |
| } |
| |
| SECTION("NaN and infinity dump as null") |
| { |
| CHECK(long_double_json(std::numeric_limits<long double>::quiet_NaN()).dump() == "null"); |
| |
| // Probe the platform's runtime behavior — `volatile` forces a runtime |
| // call rather than constexpr-folding to a known answer at compile time. |
| // Skip the infinity assertions if std::isfinite() doesn't actually |
| // recognize long double infinity on this platform (notably, Valgrind |
| // 3.22's x87 80-bit emulation reports +/-inf as a large finite value). |
| // TODO(rusloker): remove this guard once Valgrind's 80-bit long double |
| // support ships (Valgrind bug https://bugs.kde.org/show_bug.cgi?id=197915, |
| // ASSIGNED since 2009 — the Valgrind project tracks its bugs on |
| // bugs.kde.org) and the minimum supported Valgrind version contains it. |
| const volatile long double inf_probe = std::numeric_limits<long double>::infinity(); |
| if (!std::isfinite(inf_probe)) |
| { |
| CHECK(long_double_json(std::numeric_limits<long double>::infinity()).dump() == "null"); |
| CHECK(long_double_json(-std::numeric_limits<long double>::infinity()).dump() == "null"); |
| } |
| } |
| |
| SECTION("dump output matches double for exactly-representable values") |
| { |
| auto check_same = [](long double v_ld, double v_d) |
| { |
| const long_double_json j_ld = v_ld; |
| const json j_d = v_d; |
| CHECK(j_ld.dump() == j_d.dump()); |
| }; |
| |
| check_same(0.0L, 0.0); |
| check_same(0.5L, 0.5); |
| check_same(-0.5L, -0.5); |
| check_same(1.5L, 1.5); |
| check_same(-2.25L, -2.25); |
| check_same(1.0L, 1.0); |
| check_same(100.0L, 100.0); |
| } |
| } |
| |
| TEST_CASE("serialization of strings (bulk fast path)") |
| { |
| // These cases exercise the SWAR bulk-copy fast path in dump_escaped and the |
| // internal write buffer: long runs, escapes interrupting runs, 0x7F/DEL, |
| // multibyte UTF-8 under both ensure_ascii settings, and payloads larger than |
| // the write buffer. |
| |
| SECTION("long unescaped ASCII exceeds the write buffer") |
| { |
| const std::string big(3000, 'a'); |
| const json j = big; |
| CHECK(j.dump() == '"' + big + '"'); |
| CHECK(j.dump(-1, ' ', true) == '"' + big + '"'); |
| // round-trips |
| CHECK(json::parse(j.dump()) == j); |
| } |
| |
| SECTION("runs interrupted by escapes") |
| { |
| const json j = std::string(500, 'x') + "\n\"\\" + std::string(500, 'y'); |
| const std::string out = j.dump(); |
| CHECK(out == '"' + std::string(500, 'x') + "\\n\\\"\\\\" + std::string(500, 'y') + '"'); |
| CHECK(json::parse(out) == j); |
| } |
| |
| SECTION("DEL (0x7F) depends on ensure_ascii") |
| { |
| const json j = std::string("a\x7f" "b"); |
| CHECK(j.dump(-1, ' ', false) == "\"a\x7f" "b\""); // copied verbatim |
| CHECK(j.dump(-1, ' ', true) == "\"a\\u007fb\""); // escaped |
| } |
| |
| SECTION("multibyte UTF-8 under both ensure_ascii settings") |
| { |
| const json j = std::string("A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z"); // A é 你 😀 Z |
| // not escaping non-ASCII: bytes are copied through the bulk validator |
| CHECK(j.dump(-1, ' ', false) == "\"A\xc3\xa9\xe4\xbd\xa0\xf0\x9f\x98\x80Z\""); |
| // ensure_ascii: escaped (with a surrogate pair for the emoji) |
| CHECK(j.dump(-1, ' ', true) == "\"A\\u00e9\\u4f60\\ud83d\\ude00Z\""); |
| CHECK(json::parse(j.dump(-1, ' ', true)) == j); |
| } |
| |
| SECTION("many small structural writes exceed the write buffer") |
| { |
| json arr = json::array(); |
| for (int i = 0; i < 2000; ++i) |
| { |
| arr.push_back(i); |
| } |
| const std::string out = arr.dump(); |
| CHECK(out.front() == '['); |
| CHECK(out.back() == ']'); |
| CHECK(json::parse(out) == arr); |
| |
| json obj = json::object(); |
| for (int i = 0; i < 500; ++i) |
| { |
| obj["key" + std::to_string(i)] = i; |
| } |
| CHECK(json::parse(obj.dump()) == obj); |
| CHECK(json::parse(obj.dump(2)) == obj); |
| |
| // an array of many empty strings emits a long run of single-character |
| // writes ('"', '"', ',') at shallow nesting depth, so the write buffer |
| // fills and flushes mid-run without the deep recursion that would |
| // overflow the stack on some debug builds |
| json many_empty = json::array(); |
| for (int i = 0; i < 500; ++i) |
| { |
| many_empty.push_back(""); |
| } |
| const std::string out2 = many_empty.dump(); |
| CHECK(out2.size() > 1024); // spans multiple write-buffer flushes |
| CHECK(out2.front() == '['); |
| CHECK(out2.back() == ']'); |
| CHECK(json::parse(out2) == many_empty); |
| } |
| |
| SECTION("invalid UTF-8 handling is unaffected by the fast path") |
| { |
| const json j = std::string("valid\xff" "more"); |
| CHECK_THROWS_WITH_AS(utils::ignore_return_value(j.dump()), "[json.exception.type_error.316] invalid UTF-8 byte at index 5: 0xFF", json::type_error&); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::replace) == "\"valid\xef\xbf\xbd" "more\""); |
| CHECK(j.dump(-1, ' ', true, json::error_handler_t::replace) == "\"valid\\ufffdmore\""); |
| CHECK(j.dump(-1, ' ', false, json::error_handler_t::ignore) == "\"validmore\""); |
| } |
| } |
| |
| TEST_CASE("indentation is written straight into the write buffer") |
| { |
| // put_indent() memsets the indentation into the write buffer instead of |
| // copying it out of a pre-grown indentation string. These cases cover an |
| // indentation wider than the buffer, a non-space indentation character, and |
| // nesting deep enough that the accumulated indentation spans several |
| // buffer-fulls - the situations the old grow-a-string approach got wrong. |
| |
| SECTION("indent_step wider than the write buffer") |
| { |
| const json j = {{"a", 1}}; |
| // 2000 > the 1024-byte write buffer, and > the 512 the indentation |
| // string used to start at |
| CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"a\": 1\n}"); |
| // several whole buffer-fulls, so the buffer is refilled once and then |
| // flushed repeatedly |
| CHECK(j.dump(5000) == "{\n" + std::string(5000, ' ') + "\"a\": 1\n}"); |
| CHECK(j.dump(5000, '\t') == "{\n" + std::string(5000, '\t') + "\"a\": 1\n}"); |
| // an exact multiple of the buffer size |
| CHECK(j.dump(4096) == "{\n" + std::string(4096, ' ') + "\"a\": 1\n}"); |
| } |
| |
| SECTION("a non-space indentation character is used throughout") |
| { |
| const json j = {{"a", 1}}; |
| // 600 is past the point where the indentation used to be grown, which |
| // is where a hard-coded space would have shown up |
| CHECK(j.dump(600, '\t') == "{\n" + std::string(600, '\t') + "\"a\": 1\n}"); |
| CHECK(j.dump(3, '.') == "{\n...\"a\": 1\n}"); |
| } |
| |
| SECTION("accumulated indentation spans several buffer-fulls") |
| { |
| // five levels deep at 400 per level: the innermost value is indented by |
| // 2000 characters, reached in steps that each straddle the buffer end |
| json j = json::array({1}); |
| for (int i = 0; i < 4; ++i) |
| { |
| j = json::array({j}); |
| } |
| |
| const std::string out = j.dump(400); |
| CHECK(out.find(std::string("\n") + std::string(2000, ' ') + "1\n") != std::string::npos); |
| CHECK(json::parse(out) == j); |
| } |
| |
| SECTION("binary values are indented the same way") |
| { |
| // a binary value is serialized as an object with "bytes" and |
| // "subtype" keys; the byte array itself is always written compactly |
| // (see dump_byte()), so only the surrounding object's indentation |
| // goes through put_indent() |
| const json j = json::binary({1, 2, 3}, 128); |
| CHECK(j.dump(2000) == "{\n" + std::string(2000, ' ') + "\"bytes\": [1, 2, 3],\n" |
| + std::string(2000, ' ') + "\"subtype\": 128\n}"); |
| CHECK(j.dump(2000, '\t') == "{\n" + std::string(2000, '\t') + "\"bytes\": [1, 2, 3],\n" |
| + std::string(2000, '\t') + "\"subtype\": 128\n}"); |
| } |
| |
| SECTION("indentation is unchanged for ordinary widths") |
| { |
| const json j = {{"a", {1, 2}}, {"b", nullptr}}; |
| CHECK(j.dump(2) == "{\n \"a\": [\n 1,\n 2\n ],\n \"b\": null\n}"); |
| CHECK(j.dump(0) == "{\n\"a\": [\n1,\n2\n],\n\"b\": null\n}"); |
| } |
| } |
| |
| TEST_CASE("serialization of deeply nested values") |
| { |
| // dump() descends into a bounded number of levels and writes out whatever |
| // is nested deeper than that without the call stack; see |
| // https://github.com/nlohmann/json/issues/5387 |
| |
| SECTION("nested deeper than the call stack could follow") |
| { |
| // parsing is iterative, so building these costs little |
| const std::size_t depth = 100000; |
| |
| const std::string array_text = std::string(depth, '[') + '0' + std::string(depth, ']'); |
| CHECK(json::parse(array_text).dump() == array_text); |
| |
| std::string object_text; |
| object_text.reserve((6 * depth) + 1); |
| for (std::size_t i = 0; i < depth; ++i) |
| { |
| object_text += "{\"a\":"; |
| } |
| object_text += '1'; |
| object_text.append(depth, '}'); |
| CHECK(json::parse(object_text).dump() == object_text); |
| } |
| |
| SECTION("depths around the bound of the descent") |
| { |
| // Cover every depth around the bound, so that the two ways of writing a |
| // value are known to meet cleanly - wherever the bound is set. |
| for (std::size_t d = 1; d <= 300; ++d) |
| { |
| CAPTURE(d) |
| |
| const std::string array_text = std::string(d, '[') + '7' + std::string(d, ']'); |
| CHECK(json::parse(array_text).dump() == array_text); |
| |
| std::string object_text; |
| for (std::size_t i = 0; i < d; ++i) |
| { |
| object_text += "{\"k\":"; |
| } |
| object_text += '7'; |
| object_text.append(d, '}'); |
| CHECK(json::parse(object_text).dump() == object_text); |
| } |
| } |
| |
| SECTION("pretty-printing across the bound") |
| { |
| for (std::size_t d = 120; d <= 140; ++d) |
| { |
| CAPTURE(d) |
| |
| const json j = json::parse(std::string(d, '[') + '7' + std::string(d, ']')); |
| |
| std::string expected; |
| for (std::size_t i = 0; i < d; ++i) |
| { |
| expected += std::string(2 * i, ' ') + "[\n"; |
| } |
| expected += std::string(2 * d, ' ') + '7'; |
| for (std::size_t i = d; i > 0; --i) |
| { |
| expected += '\n' + std::string(2 * (i - 1), ' ') + ']'; |
| } |
| |
| CHECK(j.dump(2) == expected); |
| } |
| } |
| |
| SECTION("an empty container below the bound") |
| { |
| // an empty container is written out in full and never descended into, |
| // so it must not gain a newline when it is reached iteratively |
| for (std::size_t d = 125; d <= 135; ++d) |
| { |
| CAPTURE(d) |
| |
| const std::string compact = std::string(d, '[') + "[]" + std::string(d, ']'); |
| CHECK(json::parse(compact).dump() == compact); |
| |
| const std::string with_object = std::string(d, '[') + "{}" + std::string(d, ']'); |
| CHECK(json::parse(with_object).dump() == with_object); |
| } |
| } |
| } |
| |
| namespace |
| { |
| // wraps @a inner into @a depth single-element arrays |
| json wrap_in_arrays(const json& inner, const std::size_t depth) |
| { |
| json j = inner; |
| for (std::size_t i = 0; i < depth; ++i) |
| { |
| j = json::array({std::move(j)}); |
| } |
| return j; |
| } |
| |
| // what wrap_in_arrays(inner, depth).dump(2) is expected to be: the arrays |
| // around inner.dump(2), with inner's own lines indented by the depth |
| std::string expected_pretty_in_arrays(const json& inner, const std::size_t depth) |
| { |
| std::string expected; |
| for (std::size_t i = 0; i < depth; ++i) |
| { |
| expected += std::string(2 * i, ' ') + "[\n"; |
| } |
| |
| const std::string indent(2 * depth, ' '); |
| expected += indent; |
| for (const char c : inner.dump(2)) |
| { |
| expected += c; |
| if (c == '\n') |
| { |
| expected += indent; |
| } |
| } |
| |
| for (std::size_t i = depth; i > 0; --i) |
| { |
| expected += '\n' + std::string(2 * (i - 1), ' ') + ']'; |
| } |
| return expected; |
| } |
| } // namespace |
| |
| TEST_CASE("serialization of every kind of value below the bound of the descent") |
| { |
| // Values nested deeper than the bound are written without the call stack, |
| // by code of their own; each kind of value must come out the same there as |
| // it does at the top level, compact and pretty-printed. |
| std::vector<json> values = |
| { |
| json::parse(R"({"a": 1, "b": [1, 2, {"c": "x"}], "d": {}, "e": []})"), |
| json::parse(R"([1, [2, 3], {"k": null}, "s"])"), |
| json::object(), |
| json::array(), |
| json::binary({1, 2, 3}, 42), |
| json::binary({1, 2, 3}), |
| json::binary({}, 7), |
| json::binary({}), |
| "a string with \"escapes\"\n", |
| true, |
| false, |
| -42, |
| 42u, |
| 1.5, |
| nullptr, |
| json(json::value_t::discarded), |
| }; |
| // a pretty-printed object whose members are themselves deep |
| values.push_back({{"x", wrap_in_arrays(1, 5)}, {"y", {{"z", 2}}}}); |
| |
| for (const std::size_t depth : std::vector<std::size_t> {1, 200}) |
| { |
| CAPTURE(depth) |
| for (const auto& inner : values) |
| { |
| CAPTURE(inner.dump()) |
| const json j = wrap_in_arrays(inner, depth); |
| CHECK(j.dump() == std::string(depth, '[') + inner.dump() + std::string(depth, ']')); |
| CHECK(j.dump(2) == expected_pretty_in_arrays(inner, depth)); |
| } |
| } |
| |
| SECTION("pretty-printed objects across the bound") |
| { |
| for (std::size_t d = 120; d <= 140; ++d) |
| { |
| CAPTURE(d) |
| |
| // built from the inside out: {"k": <level below>, "n": <level>} |
| json j = 7; |
| std::string expected = "7"; |
| for (std::size_t i = d; i > 0; --i) |
| { |
| j = json({{"k", std::move(j)}, {"n", i}}); |
| |
| const std::string indent(2 * i, ' '); |
| const std::string outer_indent(2 * (i - 1), ' '); |
| std::string next = "{\n"; |
| next += indent; |
| next += "\"k\": "; |
| next += expected; |
| next += ",\n"; |
| next += indent; |
| next += "\"n\": "; |
| next += std::to_string(i); |
| next += '\n'; |
| next += outer_indent; |
| next += '}'; |
| expected = std::move(next); |
| } |
| |
| CHECK(j.dump(2) == expected); |
| CHECK(json::parse(j.dump(2)) == j); |
| CHECK(json::parse(j.dump()) == j); |
| } |
| } |
| } |
| |
| TEST_CASE("serializer buffers are flushed mid-string and mid-binary") |
| { |
| SECTION("a long run of escaped characters") |
| { |
| // each character is escaped on its own, so the escape buffer fills up |
| const json newlines = std::string(600, '\n'); |
| std::string expected = "\""; |
| for (int i = 0; i < 600; ++i) |
| { |
| expected += "\\n"; |
| } |
| expected += '"'; |
| CHECK(newlines.dump() == expected); |
| |
| // every character is \u-escaped under ensure_ascii |
| std::string umlauts; |
| std::string escaped_umlauts = "\""; |
| for (int i = 0; i < 300; ++i) |
| { |
| umlauts += "\xC3\xA4"; |
| escaped_umlauts += "\\u00e4"; |
| } |
| escaped_umlauts += '"'; |
| CHECK(json(umlauts).dump(-1, ' ', true) == escaped_umlauts); |
| } |
| |
| SECTION("a large binary value") |
| { |
| std::vector<std::uint8_t> bytes(3000); |
| std::string expected_bytes; |
| std::string expected_pretty_bytes; |
| for (std::size_t i = 0; i < bytes.size(); ++i) |
| { |
| bytes[i] = static_cast<std::uint8_t>(i % 256); |
| expected_bytes += (i == 0 ? "" : ",") + std::to_string(i % 256); |
| expected_pretty_bytes += (i == 0 ? "" : ", ") + std::to_string(i % 256); |
| } |
| const json j = json::binary(bytes); |
| CHECK(j.dump() == "{\"bytes\":[" + expected_bytes + "],\"subtype\":null}"); |
| CHECK(j.dump(2) == "{\n \"bytes\": [" + expected_pretty_bytes + "],\n \"subtype\": null\n}"); |
| } |
| } |
| |
| TEST_CASE("serialization boundary values for the write buffer") |
| { |
| // write_buffer is a std::array<char, 1024> (write_buffer_size). put_string() |
| // guards it with two checks, and each must be exercised exactly on and one |
| // past its own boundary: a heap overflow in a different manual buffer path |
| // (the dump(1100) indent buffer) once survived 100% line coverage because |
| // every test that touched it only ever grew the buffer by a single step, |
| // never landing on the exact edge of the comparison that protects it. |
| // |
| // - straight-through: put_string() bypasses write_buffer entirely and |
| // writes directly to the output adapter once `length >= write_buffer.size()`. |
| // - flush-then-copy: otherwise, if `write_buffer_pos + length > write_buffer.size()`, |
| // put_string() flushes what is pending and then memcpy's the new run into |
| // the freshly emptied buffer. |
| |
| SECTION("top-level string exercises the straight-through guard (length >= 1024)") |
| { |
| // dump() of a bare string writes the opening quote with put_char() |
| // (write_buffer_pos: 0 -> 1), then the body with put_string(). With |
| // write_buffer_pos == 1, `1 + length > 1024` and `length >= 1024` flip |
| // together at length 1024, so 1023/1024/1025 cover "just under", |
| // "exactly at" and "just over" the guard in one move: 1023 is copied |
| // into the buffer (filling it exactly), 1024 and 1025 bypass it. |
| for (const std::size_t len : |
| { |
| std::size_t{1023}, std::size_t{1024}, std::size_t{1025} |
| }) |
| { |
| CAPTURE(len) |
| const std::string body(len, 'a'); |
| const json j = body; |
| const std::string expected = '"' + body + '"'; |
| |
| CHECK(j.dump() == expected); |
| |
| std::ostringstream o; |
| o << j; |
| CHECK(o.str() == expected); |
| } |
| } |
| |
| SECTION("string nested in an array exercises the flush-then-copy guard") |
| { |
| // json::array({body}) writes '[' then '"' before the body, so |
| // write_buffer_pos == 2 when put_string() is entered for it. The |
| // body's last byte then lands at logical offset 2 + len: len == 1022 |
| // lands exactly on offset 1024 (2 + 1022 == write_buffer.size(), so the |
| // strict "> " guard does not fire and the body fits snugly), while |
| // len == 1023 lands one past it at offset 1025 (2 + 1023 > 1024), |
| // which must flush what's pending before copying the body in. |
| for (const std::size_t len : |
| { |
| std::size_t{1022}, std::size_t{1023} |
| }) |
| { |
| CAPTURE(len) |
| const std::string body(len, 'a'); |
| const json j = json::array({body}); |
| const std::string expected = "[\"" + body + "\"]"; |
| |
| CHECK(j.dump() == expected); |
| |
| std::ostringstream o; |
| o << j; |
| CHECK(o.str() == expected); |
| |
| CHECK(json::parse(j.dump()) == j); |
| } |
| } |
| } |
| |
| TEST_CASE("serialization boundary values for the string buffer") |
| { |
| // string_buffer is a std::array<char, 512>. dump_escaped_impl() flushes it |
| // mid-string once fewer than 13 bytes remain (`string_buffer.size() - bytes |
| // < 13`), 13 being one more than the most a single code point can ever |
| // write at once (a surrogate pair: two back-to-back "\uXXXX" escapes, 12 |
| // bytes). Every write into string_buffer that this check protects happens |
| // in steps of 2 (a simple "\\x" escape) or 6 (one "\uXXXX" unit), so |
| // `bytes` only ever takes even values at the point the check runs - the |
| // tightest values actually reachable are therefore 498 (512 - 498 == 14, |
| // one simple escape away from the threshold) and 500 (512 - 500 == 12, |
| // where the flush fires immediately and resets bytes to 0). |
| |
| SECTION("a run of 2-byte escapes lands bytes on, and one step past, the flush threshold") |
| { |
| for (const int count : |
| { |
| 249, 250, 251 |
| }) |
| { |
| CAPTURE(count) |
| const json j = std::string(static_cast<std::size_t>(count), '\n'); |
| std::string expected = "\""; |
| for (int i = 0; i < count; ++i) |
| { |
| expected += "\\n"; |
| } |
| expected += '"'; |
| CHECK(j.dump() == expected); |
| } |
| } |
| |
| SECTION("an ASCII prefix leaves the tightest reachable margin before a 12-byte surrogate pair") |
| { |
| // U+1F600 (the "\xF0\x9F\x98\x80" UTF-8 bytes) is dumped under |
| // ensure_ascii as the 12-byte surrogate pair "\ud83d\ude00"; that |
| // write happens in a single step with no intermediate flush check, so |
| // it is the write most exposed by an off-by-one in the "< 13" guard. |
| // A prefix of 249 newlines leaves exactly 14 bytes of headroom |
| // (512 - 498), the smallest margin the guard ever actually allows |
| // into a new code point; 250 newlines instead trigger the guard's own |
| // flush first, so the emoji starts from a freshly emptied (512-byte) |
| // buffer, and 251 repeats that with one more escape already past the |
| // reset. Together they cover the margin the guard allows landing on, |
| // one step before, and one step after - all must still produce the |
| // identical, correct escapes. |
| for (const int prefix_count : |
| { |
| 249, 250, 251 |
| }) |
| { |
| CAPTURE(prefix_count) |
| const std::string prefix(static_cast<std::size_t>(prefix_count), '\n'); |
| const std::string emoji = "\xF0\x9F\x98\x80"; |
| const json j = prefix + emoji; |
| |
| std::string expected_prefix; |
| for (int i = 0; i < prefix_count; ++i) |
| { |
| expected_prefix += "\\n"; |
| } |
| |
| // newline escaping does not depend on ensure_ascii: only the |
| // emoji differs (raw UTF-8 bytes vs. a \u-escaped surrogate pair) |
| std::string expected_raw = "\""; |
| expected_raw += expected_prefix; |
| expected_raw += emoji; |
| expected_raw += '"'; |
| std::string expected_ascii = "\""; |
| expected_ascii += expected_prefix; |
| expected_ascii += R"(\ud83d\ude00")"; |
| CHECK(j.dump(-1, ' ', false) == expected_raw); |
| CHECK(j.dump(-1, ' ', true) == expected_ascii); |
| CHECK(json::parse(j.dump(-1, ' ', true)) == j); |
| CHECK(json::parse(j.dump(-1, ' ', false)) == j); |
| } |
| } |
| |
| SECTION("SWAR bulk-copy stride: k plain bytes followed by a byte handled individually") |
| { |
| // string_bulk_run()/find_ascii_copyable_run() (string_scan.hpp) scan 8 |
| // bytes at a time and fall back to a byte-at-a-time tail scan for |
| // what is left over. k from 0 to 17 spans zero, one and two full |
| // 8-byte strides plus a 1-byte tail, so every possible stopping point |
| // within and right after the SIMD stride is covered. |
| for (std::size_t k = 0; k <= 17; ++k) |
| { |
| CAPTURE(k) |
| const std::string prefix(k, 'a'); |
| |
| // (a) the run is stopped by a quote that must itself be escaped |
| { |
| const json j = prefix + "\""; |
| CHECK(j.dump() == '"' + prefix + "\\\"" + '"'); |
| } |
| |
| // (b) the run is stopped by a control character |
| { |
| const json j = prefix + "\x01"; |
| CHECK(j.dump() == '"' + prefix + "\\u0001" + '"'); |
| } |
| |
| // (c) the run is stopped by a non-ASCII byte under ensure_ascii |
| { |
| const json j = prefix + "\xC3\xA9"; // prefix + 'é' |
| CHECK(j.dump(-1, ' ', true) == '"' + prefix + "\\u00e9" + '"'); |
| } |
| } |
| } |
| } |