Compare commits

..
Author SHA1 Message Date
Niels Lohmann 6b43220a50 Reject malformed UTF-16/UTF-32 units in wide-string input
The wide-string input adapter (used for std::u16string, std::u32string,
std::wstring, and iterators over 2- or 4-byte character types) passed
some malformed code units on to the lexer as values that are neither a
byte (0x00..0xFF) nor char_traits<char>::eof(). As a result:

- A lone UTF-16 surrogate inside true/false/null was accepted if its low
  byte matched the expected letter, or ended the input silently if it
  was the last unit.
- A high surrogate followed by a unit that is not its low surrogate
  swallowed that unit; if the swallowed unit was the newline ending a
  // comment, the comment silently extended over the next line.
- Where wint_t is a signed int (macOS, the BSDs), a negative wchar_t
  collided with char_traits<char>::eof() (ending the input early) or was
  truncated to its low byte, depending on its value.

The UTF-32 helper now converts the code unit to std::uint32_t before the
range checks, so a negative unit reaches the same "emit 0xFF" branch
already used for code points above U+10FFFF. The UTF-16 helper now
peeks at the next unit before consuming it, and emits 0xFF instead of
the raw surrogate when no valid pair is found, matching how ill-formed
UTF-8 bytes are rejected elsewhere in the lexer.

Fixes #5645.
Signed-off-by: Niels Lohmann <mail@nlohmann.me>
2026-09-29 23:45:24 +02:00
19 changed files with 134 additions and 188 deletions
-35
View File
@@ -71,38 +71,3 @@ jobs:
run: cmake --build build --parallel 10
- name: Test
run: cd build ; ctest -j 10 --output-on-failure
swiftpm:
runs-on: macos-15
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
persist-credentials: false
- name: Check that Package.swift resolves without a deprecation warning
run: swift package dump-package
- name: Build the SwiftPM documentation example against this checkout
run: |
mkdir -p /tmp/json-swiftpm-consumer/Sources/MyLibrary
cp docs/mkdocs/docs/integration/swift/example.cpp /tmp/json-swiftpm-consumer/Sources/MyLibrary/example.cpp
cat > /tmp/json-swiftpm-consumer/Package.swift << EOF
// swift-tools-version: 5.9
import PackageDescription
let package = Package(
name: "MyPackage",
dependencies: [
.package(path: "${{ github.workspace }}")
],
targets: [
.target(
name: "MyLibrary",
dependencies: [
.product(name: "json", package: "json")
],
publicHeadersPath: "."
)
]
)
EOF
cd /tmp/json-swiftpm-consumer
swift build
+40
View File
@@ -0,0 +1,40 @@
Format: https://www.debian.org/doc/packaging-manuals/copyright-format/1.0/
Upstream-Name: json
Upstream-Contact: Niels Lohmann <mail@nlohmann.me>
Source: https://github.com/nlohmann/json
Files: *
Copyright: 2013-2026 Niels Lohmann <https://nlohmann.me>
License: MIT
Files: include/nlohmann/thirdparty/hedley.hpp
Copyright: 2016-2021 Evan Nemerson <evan@nemerson.com>
License: CC0
Files: include/nlohmann/detail/meta/cpp_future.hpp
Copyright: 2013-2026 Niels Lohmann <https://nlohmann.me> and 2018 The Abseil Authors
License: MIT AND Apache-2.0
Files: tests/thirdparty/doctest/*
Copyright: 2016-2023 Viktor Kirilov
License: MIT
Files: tests/thirdparty/fifo_map/*
Copyright: 2015-2017 Niels Lohmann
License: MIT
Files: tests/thirdparty/Fuzzer/*
Copyright: 2003-2022 LLVM Project.
License: Apache-2.0
Files: tests/thirdparty/imapdl/*
Copyright: 2017 Georg Sauthoff <mail@gms.tf>
License: GPL-3.0-only
Files: tools/amalgamate/*
Copyright: 2012 Erik Edlund <erik.edlund@32767.se>
License: BSD-3-Clause
Files: tools/gdb_pretty_printer/*
Copyright: 2020 Hannes Domani <https://github.com/ssbssa>
License: MIT
-1
View File
@@ -77,7 +77,6 @@ cc_library(
name = "singleheader-json",
hdrs = [
"single_include/nlohmann/json.hpp",
"single_include/nlohmann/json_fwd.hpp",
],
includes = ["single_include"],
visibility = ["//visibility:public"],
+1 -1
View File
@@ -10,5 +10,5 @@ title: "JSON for Modern C++"
version: 3.12.0
date-released: 2025-04-07
license: MIT
repository-code: "https://github.com/nlohmann/json"
repository-code: "https://github.com/nlohmann"
url: https://json.nlohmann.me
+1 -4
View File
@@ -42,11 +42,8 @@ endif()
## OPTIONS
##
# Build the tests by default only for the main project and only if the tests
# directory exists (the release archive json.tar.xz does not contain it).
# VERSION_GREATER_EQUAL is not available in older CMake (< 3.7)
if(${MAIN_PROJECT} AND (${CMAKE_VERSION} VERSION_EQUAL 3.13 OR ${CMAKE_VERSION} VERSION_GREATER 3.13)
AND EXISTS "${CMAKE_CURRENT_SOURCE_DIR}/tests/CMakeLists.txt")
if(${MAIN_PROJECT} AND (${CMAKE_VERSION} VERSION_EQUAL 3.13 OR ${CMAKE_VERSION} VERSION_GREATER 3.13))
set(JSON_BuildTests_INIT ON)
else()
set(JSON_BuildTests_INIT OFF)
+3 -3
View File
@@ -207,19 +207,19 @@ Further documentation:
## REUSE
### `REUSE.toml`
### `.reuse/dep5`
The file defines the licenses of certain third-party components in the repository. The root `Makefile` contains a target `reuse` that checks for compliance.
Further documentation:
- [REUSE.toml](https://reuse.software/spec-3.3/#reusetoml)
- [DEP5](https://reuse.software/spec-3.2/#dep5-deprecated)
- [reuse command-line tool](https://pypi.org/project/reuse/)
- [documentation of linting](https://reuse.readthedocs.io/en/stable/man/reuse-lint.html)
- [REUSE](http://reuse.software)
> [!IMPORTANT]
> The filename `REUSE.toml` is predetermined by REUSE. Alternatively, a `.reuse/dep5` file (deprecated) can be used.
> The filename `.reuse/dep5` is predetermined by REUSE. Alternatively, a `REUSE.toml` file can be used.
### `.reuse/templates`
+2 -5
View File
@@ -7,9 +7,6 @@
# find GNU sed to use `-i` parameter
SED:=$(shell command -v gsed || which sed)
# find GNU tar to use `--sort` and `--pax-option` parameters
TAR:=$(shell command -v gtar || which tar)
##########################################################################
# source files
@@ -218,8 +215,8 @@ ChangeLog.md:
# archive is created according to the advices of <https://reproducible-builds.org/docs/archives/>.
json.tar.xz:
mkdir json
rsync -R $(shell find LICENSE.MIT nlohmann_json.natvis CMakeLists.txt cmake/*.in include single_include src/modules -type f) json
$(TAR) --sort=name --mtime="@$(shell git log -1 --pretty=%ct)" --owner=0 --group=0 --numeric-owner --pax-option=exthdr.name=%d/PaxHeaders/%f,delete=atime,delete=ctime --create --file - json | xz --compress -9e --threads=2 - > json.tar.xz
rsync -R $(shell find LICENSE.MIT nlohmann_json.natvis CMakeLists.txt cmake/*.in include single_include -type f) json
gtar --sort=name --mtime="@$(shell git log -1 --pretty=%ct)" --owner=0 --group=0 --numeric-owner --pax-option=exthdr.name=%d/PaxHeaders/%f,delete=atime,delete=ctime --create --file - json | xz --compress -9e --threads=2 - > json.tar.xz
rm -fr json
# We use `-X` to make the resulting ZIP file reproducible, see
+1 -1
View File
@@ -6,7 +6,7 @@ import PackageDescription
let package = Package(
name: "nlohmann-json",
platforms: [
.iOS(.v12), .macOS(.v10_13), .tvOS(.v12), .watchOS(.v9), .visionOS(.v1)
.iOS(.v12), .macOS(.v10_13), .tvOS(.v12), .watchOS(.v4), .visionOS(.v1)
],
products: [
.library(name: "json", targets: ["json"])
+10 -13
View File
@@ -1204,18 +1204,15 @@ language bindings, format converters, and the like. See the curated [Ecosystem](
Though it's 2026 already, the support for C++11 is still a bit sparse. Currently, the following compilers are known to work:
- GCC 4.8 - 16.2 (and possibly later)
- Clang 3.4 - 22.1 (and possibly later)
- Apple Clang 15.0 - 21.0 (and possibly later)
- Intel C++ Compiler Classic (icpc) 2021.10
- Intel oneAPI DPC++/C++ Compiler (icpx) 2025.3 (and possibly later)
- NVIDIA CUDA Compiler (nvcc) 11.8 - 12.6 (and possibly later)
- NVIDIA HPC SDK C++ Compiler (nvc++) 25.5 (and possibly later)
- Microsoft Visual C++ 2015 / MSVC 19.0 (and possibly later)
- Microsoft Visual C++ 2017 / MSVC 19.16 (and possibly later)
- Microsoft Visual C++ 2019 / MSVC 19.29 (and possibly later)
- Microsoft Visual C++ 2022 / MSVC 19.44 (and possibly later)
- Microsoft Visual C++ 2026 / MSVC 19.51 (and possibly later)
- GCC 4.8 - 14.2 (and possibly later)
- Clang 3.4 - 21.0 (and possibly later)
- Apple Clang 9.1 - 16.0 (and possibly later)
- Intel C++ Compiler 17.0.2 (and possibly later)
- Nvidia CUDA Compiler 11.0.221 (and possibly later)
- Microsoft Visual C++ 2015 / Build Tools 14.0.25123.0 (and possibly later)
- Microsoft Visual C++ 2017 / Build Tools 15.5.180.51428 (and possibly later)
- Microsoft Visual C++ 2019 / Build Tools 16.3.1+1def00d3d (and possibly later)
- Microsoft Visual C++ 2022 / Build Tools 19.30.30709.0 (and possibly later)
I would be happy to learn about other compilers/versions.
@@ -1405,7 +1402,7 @@ The library is compliant to version 3.3 of the [**REUSE specification**](https:/
- Every source file contains an SPDX copyright header.
- The full text of all licenses used in the repository can be found in the `LICENSES` folder.
- File `REUSE.toml` contains an overview of all files' copyrights and licenses.
- File `.reuse/dep5` contains an overview of all files' copyrights and licenses.
- Run `pipx run reuse lint` to verify the project's REUSE compliance and `pipx run reuse spdx` to generate a SPDX SBOM.
## Contact
-58
View File
@@ -1,58 +0,0 @@
version = 1
SPDX-PackageName = "json"
SPDX-PackageSupplier = "Niels Lohmann <mail@nlohmann.me>"
SPDX-PackageDownloadLocation = "https://github.com/nlohmann/json"
[[annotations]]
path = "**"
precedence = "aggregate"
SPDX-FileCopyrightText = "2013-2026 Niels Lohmann <https://nlohmann.me>"
SPDX-License-Identifier = "MIT"
[[annotations]]
path = "include/nlohmann/thirdparty/hedley.hpp"
precedence = "aggregate"
SPDX-FileCopyrightText = "2016-2021 Evan Nemerson <evan@nemerson.com>"
SPDX-License-Identifier = "CC0"
[[annotations]]
path = "include/nlohmann/detail/meta/cpp_future.hpp"
precedence = "aggregate"
SPDX-FileCopyrightText = "2013-2026 Niels Lohmann <https://nlohmann.me> and 2018 The Abseil Authors"
SPDX-License-Identifier = "MIT AND Apache-2.0"
[[annotations]]
path = "tests/thirdparty/doctest/**"
precedence = "aggregate"
SPDX-FileCopyrightText = "2016-2023 Viktor Kirilov"
SPDX-License-Identifier = "MIT"
[[annotations]]
path = "tests/thirdparty/fifo_map/**"
precedence = "aggregate"
SPDX-FileCopyrightText = "2015-2017 Niels Lohmann"
SPDX-License-Identifier = "MIT"
[[annotations]]
path = "tests/thirdparty/Fuzzer/**"
precedence = "aggregate"
SPDX-FileCopyrightText = "2003-2022 LLVM Project."
SPDX-License-Identifier = "Apache-2.0"
[[annotations]]
path = "tests/thirdparty/imapdl/**"
precedence = "aggregate"
SPDX-FileCopyrightText = "2017 Georg Sauthoff <mail@gms.tf>"
SPDX-License-Identifier = "GPL-3.0-only"
[[annotations]]
path = "tools/amalgamate/**"
precedence = "aggregate"
SPDX-FileCopyrightText = "2012 Erik Edlund <erik.edlund@32767.se>"
SPDX-License-Identifier = "BSD-3-Clause"
[[annotations]]
path = "tools/gdb_pretty_printer/**"
precedence = "aggregate"
SPDX-FileCopyrightText = "2020 Hannes Domani <https://github.com/ssbssa>"
SPDX-License-Identifier = "MIT"
-1
View File
@@ -48,7 +48,6 @@ cc_library(
name = "singleheader-json",
hdrs = [
"single_include/nlohmann/json.hpp",
"single_include/nlohmann/json_fwd.hpp",
],
includes = ["single_include"],
visibility = ["//visibility:public"],
+1 -1
View File
@@ -64,7 +64,7 @@ if(MODE STREQUAL "undef")
# recipe is self-contained and its output is byte-stable across reruns.
# The embedded SPDX tags below are part of the *generated* file's
# content, not a REUSE header for this .cmake script itself (which is
# already covered by the blanket path = "**" rule in REUSE.toml) -- keep
# already covered by the blanket "Files: *" rule in .reuse/dep5) -- keep
# them wrapped in REUSE-IgnoreStart/End so `reuse lint` does not try to
# parse "MIT\n")" as this file's own SPDX-License-Identifier value.
# REUSE-IgnoreStart
+1 -1
View File
@@ -125,7 +125,7 @@ automatically download a release as a dependency at configure time.
### `JSON_BuildTests`
Build the unit tests when [`BUILD_TESTING`](https://cmake.org/cmake/help/latest/command/enable_testing.html) is enabled. This option is `ON` by default if the library's CMake project is the top project and the `tests` directory exists (the release archive `json.tar.xz` does not contain it). That is, when integrating the library as described above, the test suite is not built unless explicitly switched on with this option.
Build the unit tests when [`BUILD_TESTING`](https://cmake.org/cmake/help/latest/command/enable_testing.html) is enabled. This option is `ON` by default if the library's CMake project is the top project. That is, when integrating the library as described above, the test suite is not built unless explicitly switched on with this option.
### `JSON_CI`
@@ -443,30 +443,6 @@ installed by adding the `-DJSON_MultipleHeaders=ON` flag (i.e., `cget install nl
- :octicons-file-24: File issues at the [library issue tracker](https://github.com/nlohmann/json/issues)
- :octicons-question-24: [Xcode documentation](https://developer.apple.com/documentation/xcode/adding-package-dependencies-to-your-app)
The `json` target's public headers live at `single_include/nlohmann`, so a consumer must write `#include <json.hpp>` rather than the
`#include <nlohmann/json.hpp>` form used elsewhere in this documentation. The `json` target also ships only headers, and SwiftPM/Xcode
expect every library target to produce an object file to link against; without one, linking a consumer fails with a missing `json.o`
([#4650](https://github.com/nlohmann/json/issues/4650)). The workaround is to add at least one `.cpp` file of your own to the target
that depends on `json`.
??? example
1. Create the following files:
```swift title="Package.swift"
--8<-- "integration/swift/Package.swift"
```
```cpp title="Sources/MyLibrary/example.cpp"
--8<-- "integration/swift/example.cpp"
```
2. Build
```shell
swift build
```
## NuGet
!!! abstract "Summary"
@@ -1,20 +0,0 @@
// swift-tools-version: 5.9
import PackageDescription
let package = Package(
name: "MyPackage",
dependencies: [
.package(url: "https://github.com/nlohmann/json.git", from: "3.12.0")
],
targets: [
// the C++ target that uses nlohmann/json
.target(
name: "MyLibrary",
dependencies: [
.product(name: "json", package: "json")
],
// works around missing public headers in MyLibrary; not related to nlohmann/json
publicHeadersPath: "."
)
]
)
@@ -1,8 +0,0 @@
// MyLibrary must contain at least one .cpp file, or SwiftPM/Xcode will
// not build a usable "json" library to link against (see nlohmann/json#4650)
#include <json.hpp>
nlohmann::json example()
{
return nlohmann::json::meta();
}
@@ -445,8 +445,10 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
}
else
{
// get the current character
const auto wc = input.get_character();
// get the current character; converted to an unsigned type so that
// a negative unit (wint_t is signed on some platforms) is not
// mistaken for an ASCII character or for EOF
const auto wc = static_cast<std::uint32_t>(input.get_character());
// UTF-32 to UTF-8 encoding
if (wc < 0x80)
@@ -541,9 +543,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
bool valid_pair = false;
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
{
const auto wc2 = static_cast<unsigned int>(input.get_character());
// only consume the next unit if it completes the pair
const auto wc2 = static_cast<unsigned int>(*input.current);
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
input.get_character();
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
@@ -556,7 +560,8 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (!valid_pair)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
utf8_bytes[0] = 0xFF;
utf8_bytes_filled = 1;
}
}
+9 -4
View File
@@ -7989,8 +7989,10 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
}
else
{
// get the current character
const auto wc = input.get_character();
// get the current character; converted to an unsigned type so that
// a negative unit (wint_t is signed on some platforms) is not
// mistaken for an ASCII character or for EOF
const auto wc = static_cast<std::uint32_t>(input.get_character());
// UTF-32 to UTF-8 encoding
if (wc < 0x80)
@@ -8085,9 +8087,11 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
bool valid_pair = false;
if (wc <= 0xDBFF && JSON_HEDLEY_UNLIKELY(!input.empty()))
{
const auto wc2 = static_cast<unsigned int>(input.get_character());
// only consume the next unit if it completes the pair
const auto wc2 = static_cast<unsigned int>(*input.current);
if (0xDC00 <= wc2 && wc2 <= 0xDFFF)
{
input.get_character();
const auto charcode = 0x10000u + (((static_cast<unsigned int>(wc) & 0x3FFu) << 10u) | (wc2 & 0x3FFu));
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(0xF0u | (charcode >> 18u));
utf8_bytes[1] = static_cast<std::char_traits<char>::int_type>(0x80u | ((charcode >> 12u) & 0x3Fu));
@@ -8100,7 +8104,8 @@ struct wide_string_input_helper<BaseInputAdapter, 2>
if (!valid_pair)
{
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
// emit a byte that is never valid UTF-8 (see the UTF-32 case)
utf8_bytes[0] = 0xFF;
utf8_bytes_filled = 1;
}
}
+56 -4
View File
@@ -8,6 +8,7 @@
#include "doctest_compatibility.h"
#include <cwchar>
#include <nlohmann/json.hpp>
using nlohmann::json;
@@ -98,15 +99,15 @@ TEST_CASE("wide strings")
CHECK_THROWS_AS(_ = json::parse(w), json::parse_error&);
// a lone low surrogate cannot start a pair
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// a high surrogate followed by a non-low-surrogate unit is invalid
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// ... also when the unit is above the low surrogates
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xD800, 0xE000, u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// a lone low surrogate must not swallow the following unit: pairing
// it with any second unit would produce valid UTF-8, so the error
// has to report an ill-formed byte at the surrogate's own position
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"<U+0000>'", json::parse_error&);
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'"', 0xDC00, u'a', u'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
// a valid surrogate pair is still decoded (U+1F600)
CHECK(json::parse(std::u16string{u'"', 0xD83D, 0xDE00, u'"'}).get<std::string>() == "\xF0\x9F\x98\x80");
}
@@ -141,5 +142,56 @@ TEST_CASE("wide strings")
CHECK_THROWS_WITH_AS(_ = json::parse(std::u32string{U'"', static_cast<char32_t>(0xFFFFFFFF), U'"'}), "[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid string: ill-formed UTF-8 byte; last read: '\"\xFF'", json::parse_error&);
}
}
SECTION("malformed wide-string input outside strings (#5645)")
{
json _;
// a lone low surrogate inside a literal must not be truncated to its
// low byte and mistaken for the letter the literal expects next
// (0xDC72 truncates to 'r', which is what "true" expects after 't')
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xDC72), u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... also when the lone surrogate is the last unit of the input
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u'f', u'a', u'l', u's', static_cast<char16_t>(0xDD65)}),
"[json.exception.parse_error.101] parse error at line 1, column 5: syntax error while parsing value - invalid literal; last read: 'fals\xFF'", json::parse_error&);
// a high surrogate followed by a unit that is not its low surrogate
// must not silently swallow that unit
CHECK_THROWS_WITH_AS(_ = json::parse(std::u16string{u't', static_cast<char16_t>(0xD872), u'X', u'u', u'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
// ... in particular, if the swallowed unit is the newline that ends a
// // comment, the comment must not extend over the following line
CHECK(json::parse(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'},
nullptr, true, /*ignore_comments*/true) == json::parse("[1,2]"));
CHECK(json::accept(std::u16string{u'[', u'1', u' ', u'/', u'/', static_cast<char16_t>(0xD800), u'\n',
u',', u'2', u' ', u'/', u'/', u'\n', u']'}, /*ignore_comments*/true));
// cases 5 and 6 use a 32-bit wchar_t (Linux, macOS, the BSDs) to reach
// the UTF-32 helper tested above via u32string; the 16-bit wchar_t of
// Windows goes through the UTF-16 helper instead, already covered by
// the u16string cases above
#if WCHAR_MAX > 0xFFFFu
// a negative wchar_t must not be mistaken for
// char_traits<char>::eof() and silently end the input, letting
// trailing garbage pass the strict end-of-input check (only observable
// where wint_t is signed, e.g. macOS/the BSDs; on Linux wint_t is
// unsigned and this was already handled by #5348)
std::wstring w = L"[1]";
w.push_back(static_cast<wchar_t>(-1));
w += L"garbage";
CHECK(!json::accept(w));
CHECK_THROWS_WITH_AS(_ = json::parse(w),
"[json.exception.parse_error.101] parse error at line 1, column 4: syntax error while parsing value - invalid literal; last read: '1]\xFF'; expected end of input", json::parse_error&);
// other negative wchar_t units must not be truncated to their low
// byte (0xFFFFFF72 truncates to 'r', as in the u16string case above)
CHECK_THROWS_WITH_AS(_ = json::parse(std::wstring{L't', static_cast<wchar_t>(0xFFFFFF72), L'u', L'e'}),
"[json.exception.parse_error.101] parse error at line 1, column 2: syntax error while parsing value - invalid literal; last read: 't\xFF'", json::parse_error&);
#endif
}
}
#endif