mirror of
https://github.com/nlohmann/json.git
synced 2026-10-01 20:20:32 +00:00
Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
11e75a5428 | ||
|
|
e400780533 |
@@ -0,0 +1,53 @@
|
|||||||
|
name: "Cancel runs of closed pull requests"
|
||||||
|
|
||||||
|
# The concurrency groups of the other workflows cancel superseded runs when a
|
||||||
|
# pull request gets new commits, but nothing stops the runs of its last commit
|
||||||
|
# once the pull request is merged or closed. They then keep the runners busy
|
||||||
|
# for hours while the queue of the open pull requests waits.
|
||||||
|
#
|
||||||
|
# pull_request_target is needed to get a token that can cancel runs for pull
|
||||||
|
# requests from forks. This is safe because the workflow never checks out or
|
||||||
|
# runs code from the pull request; it only calls the API.
|
||||||
|
on:
|
||||||
|
pull_request_target:
|
||||||
|
types: [closed]
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.run_id }}
|
||||||
|
cancel-in-progress: true
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
cancel:
|
||||||
|
permissions:
|
||||||
|
actions: write
|
||||||
|
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- name: Harden Runner
|
||||||
|
uses: step-security/harden-runner@e14015d583714f6e62063499dc959a02595150a1 # v2.21.1
|
||||||
|
with:
|
||||||
|
egress-policy: audit
|
||||||
|
|
||||||
|
- name: Cancel unfinished runs of the pull request's head commit
|
||||||
|
env:
|
||||||
|
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
GH_REPO: ${{ github.repository }}
|
||||||
|
HEAD_SHA: ${{ github.event.pull_request.head.sha }}
|
||||||
|
SELF: ${{ github.run_id }}
|
||||||
|
# only runs triggered by the pull request: when a branch is pushed to
|
||||||
|
# develop directly, its push runs share the head commit
|
||||||
|
run: |
|
||||||
|
gh api --paginate "repos/$GH_REPO/actions/runs?head_sha=$HEAD_SHA&per_page=100" \
|
||||||
|
--jq ".workflow_runs[]
|
||||||
|
| select(.status != \"completed\" and .id != $SELF)
|
||||||
|
| select(.event == \"pull_request\" or .event == \"pull_request_target\")
|
||||||
|
| \"\(.id) \(.name)\"" |
|
||||||
|
while read -r id name; do
|
||||||
|
echo "Cancelling run $id ($name)"
|
||||||
|
# a run may finish between listing and cancelling; that is not an error
|
||||||
|
gh run cancel "$id" || true
|
||||||
|
done
|
||||||
@@ -2,16 +2,23 @@ name: "Check amalgamation"
|
|||||||
|
|
||||||
on:
|
on:
|
||||||
pull_request:
|
pull_request:
|
||||||
|
# also check develop itself: a PR can be merged before its own run of this
|
||||||
|
# workflow completes (e.g. while it is still queued), leaving single_include
|
||||||
|
# stale on develop without any failing check
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- develop
|
||||||
|
|
||||||
concurrency:
|
concurrency:
|
||||||
group: ${{ github.workflow }}-${{ github.ref || github.run_id }}
|
group: ${{ github.workflow }}-${{ github.ref || github.run_id }}
|
||||||
cancel-in-progress: true
|
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||||
|
|
||||||
permissions:
|
permissions:
|
||||||
contents: read
|
contents: read
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
save:
|
save:
|
||||||
|
if: github.event_name == 'pull_request'
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- name: Harden Runner
|
- name: Harden Runner
|
||||||
@@ -43,11 +50,11 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
egress-policy: audit
|
egress-policy: audit
|
||||||
|
|
||||||
- name: Checkout pull request
|
- name: Checkout pull request or pushed commit
|
||||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||||
with:
|
with:
|
||||||
path: main
|
path: main
|
||||||
ref: ${{ github.event.pull_request.head.sha }}
|
ref: ${{ github.event.pull_request.head.sha || github.sha }}
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
|
|
||||||
- name: Checkout tools
|
- name: Checkout tools
|
||||||
|
|||||||
@@ -10,7 +10,8 @@ permissions:
|
|||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
comment:
|
comment:
|
||||||
if: ${{ github.event.workflow_run.conclusion == 'failure' }}
|
# push runs on develop have no PR to comment on (and no "pr" artifact)
|
||||||
|
if: ${{ github.event.workflow_run.conclusion == 'failure' && github.event.workflow_run.event == 'pull_request' }}
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
permissions:
|
permissions:
|
||||||
contents: read
|
contents: read
|
||||||
|
|||||||
@@ -56,8 +56,6 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
|
|||||||
|
|
||||||
- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size`
|
- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size`
|
||||||
is false.
|
is false.
|
||||||
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is
|
|
||||||
not valid UTF-8
|
|
||||||
|
|
||||||
## Complexity
|
## Complexity
|
||||||
|
|
||||||
@@ -91,5 +89,4 @@ Linear in the size of the JSON value `j`.
|
|||||||
## Version history
|
## Version history
|
||||||
|
|
||||||
- Added in version 3.11.0.
|
- Added in version 3.11.0.
|
||||||
- BJData version parameter (for draft3 binary encoding) added in version 3.12.0.
|
- BJData version parameter (for draft3 binary encoding) added in version 3.12.0.
|
||||||
- Throwing `type_error.316` for a string or object key that is not valid UTF-8 added in version 3.13.0.
|
|
||||||
@@ -46,8 +46,6 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
|
|||||||
- Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value
|
- Throws [`out_of_range.415`](../../home/exceptions.md#jsonexceptionout_of_range415) if the subtype of a binary value
|
||||||
exceeds 255, the maximum of the BSON binary subtype; example:
|
exceeds 255, the maximum of the BSON binary subtype; example:
|
||||||
`"subtype 70000 is too large for the BSON binary subtype (max 255)"`
|
`"subtype 70000 is too large for the BSON binary subtype (max 255)"`
|
||||||
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key is
|
|
||||||
not valid UTF-8
|
|
||||||
|
|
||||||
## Complexity
|
## Complexity
|
||||||
|
|
||||||
@@ -84,5 +82,3 @@ pass before anything is written.
|
|||||||
- Added in version 3.4.0.
|
- Added in version 3.4.0.
|
||||||
- Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0.
|
- Linear in the size of `j`, and no longer limited by the call stack for deeply nested values, since version 3.13.0.
|
||||||
- `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0.
|
- `out_of_range.415` is now detected before anything is written, like the other exceptions above, since version 3.13.0.
|
||||||
- Throwing `type_error.316` for a string value or object key that is not valid UTF-8, detected before anything is
|
|
||||||
written, added in version 3.13.0.
|
|
||||||
|
|||||||
@@ -35,11 +35,6 @@ The exact mapping and its limitations are described on a [dedicated page](../../
|
|||||||
|
|
||||||
Strong guarantee: if an exception is thrown, there are no changes in the JSON value.
|
Strong guarantee: if an exception is thrown, there are no changes in the JSON value.
|
||||||
|
|
||||||
## Exceptions
|
|
||||||
|
|
||||||
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is
|
|
||||||
not valid UTF-8
|
|
||||||
|
|
||||||
## Complexity
|
## Complexity
|
||||||
|
|
||||||
Linear in the size of the JSON value `j`.
|
Linear in the size of the JSON value `j`.
|
||||||
@@ -73,4 +68,3 @@ Linear in the size of the JSON value `j`.
|
|||||||
|
|
||||||
- Added in version 2.0.9.
|
- Added in version 2.0.9.
|
||||||
- Compact representation of floating-point numbers added in version 3.8.0.
|
- Compact representation of floating-point numbers added in version 3.8.0.
|
||||||
- Throwing `type_error.316` for a string or object key that is not valid UTF-8 added in version 3.13.0.
|
|
||||||
|
|||||||
@@ -49,8 +49,6 @@ Strong guarantee: if an exception is thrown, there are no changes in the JSON va
|
|||||||
|
|
||||||
- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size`
|
- Throws [`other_error.502`](../../home/exceptions.md#jsonexceptionother_error502) if `use_type` is true and `use_size`
|
||||||
is false.
|
is false.
|
||||||
- Throws [type_error.316](../../home/exceptions.md#jsonexceptiontype_error316) if a string or object key in `j` is
|
|
||||||
not valid UTF-8
|
|
||||||
|
|
||||||
## Complexity
|
## Complexity
|
||||||
|
|
||||||
@@ -84,4 +82,3 @@ Linear in the size of the JSON value `j`.
|
|||||||
## Version history
|
## Version history
|
||||||
|
|
||||||
- Added in version 3.1.0.
|
- Added in version 3.1.0.
|
||||||
- Throwing `type_error.316` for a string or object key that is not valid UTF-8 added in version 3.13.0.
|
|
||||||
|
|||||||
@@ -63,12 +63,6 @@ The library uses the following mapping from JSON values types to BJData types ac
|
|||||||
|
|
||||||
- strings with more than 18446744073709551615 bytes, i.e., 2<sup>64</sup>-1 bytes (theoretical)
|
- strings with more than 18446744073709551615 bytes, i.e., 2<sup>64</sup>-1 bytes (theoretical)
|
||||||
|
|
||||||
!!! warning "UTF-8 validation of string values and object keys"
|
|
||||||
|
|
||||||
BJData strings must use UTF-8 encoding. `to_bjdata()` validates the bytes of every string value and object key
|
|
||||||
and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, so a
|
|
||||||
value with such a string cannot be serialized in the first place.
|
|
||||||
|
|
||||||
!!! info "Unused BJData markers"
|
!!! info "Unused BJData markers"
|
||||||
|
|
||||||
The following markers are not used in the conversion:
|
The following markers are not used in the conversion:
|
||||||
@@ -214,15 +208,6 @@ The library maps BJData types to JSON value types as follows:
|
|||||||
|
|
||||||
The mapping is **complete** in the sense that any BJData value can be converted to a JSON value.
|
The mapping is **complete** in the sense that any BJData value can be converted to a JSON value.
|
||||||
|
|
||||||
!!! warning "Ill-formed UTF-8 in string values and object keys"
|
|
||||||
|
|
||||||
BJData strings must use UTF-8 encoding, but this is not enforced on read: `from_bjdata()` accepts a string
|
|
||||||
value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However,
|
|
||||||
[`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws
|
|
||||||
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error
|
|
||||||
handler is passed that replaces or ignores the ill-formed bytes. `to_bjdata()` is strict as well (see above), so
|
|
||||||
a value read this way cannot be written back to BJData.
|
|
||||||
|
|
||||||
!!! info "Round trips"
|
!!! info "Round trips"
|
||||||
|
|
||||||
A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with
|
A value returned by [`from_bjdata`](../../api/basic_json/from_bjdata.md) can be serialized with
|
||||||
|
|||||||
@@ -109,17 +109,14 @@ The library maps BSON record types to JSON value types as follows:
|
|||||||
If BSON input must be validated for strict specification compliance, validate it separately before passing it to
|
If BSON input must be validated for strict specification compliance, validate it separately before passing it to
|
||||||
`from_bson()`.
|
`from_bson()`.
|
||||||
|
|
||||||
!!! warning "Ill-formed UTF-8 in string values"
|
!!! warning "UTF-8 validation of string values"
|
||||||
|
|
||||||
The BSON specification requires `string` values (type `0x02`) to be valid UTF-8, but this is not required of a
|
The BSON specification requires `string` values (type `0x02`) to be valid UTF-8. This library validates the
|
||||||
decoder. `from_bson()` accepts a `string` value whose bytes are not valid UTF-8 and hands them back unchanged.
|
bytes of every such string at decode time and rejects ill-formed UTF-8 with a
|
||||||
However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws
|
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with `allow_exceptions`
|
||||||
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error
|
set to `false`, a discarded value), rather than only failing later when the resulting value is dumped. Element
|
||||||
handler is passed that replaces or ignores the ill-formed bytes. `to_bson()` is strict as well and throws the
|
(key) names and `binary` values (type `0x05`) are unaffected and are never validated, since they are read
|
||||||
same exception for a string value or element (key) name that is not valid UTF-8, so an object with such a key
|
byte-by-byte as a C string, or are not required to hold text, respectively.
|
||||||
or value cannot be produced in the first place, even though `from_bson()` would accept it from another source.
|
|
||||||
Element (key) names are never validated on read, since they are read byte-by-byte as a C string. `binary`
|
|
||||||
values (type `0x05`) are unaffected, since they are not required to hold text.
|
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
|
|
||||||
|
|||||||
@@ -189,16 +189,15 @@ The library maps CBOR types to JSON value types as follows:
|
|||||||
([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a
|
([RFC 8392](https://www.rfc-editor.org/rfc/rfc8392.html)), cannot be read with this library and need a
|
||||||
general-purpose CBOR library instead.
|
general-purpose CBOR library instead.
|
||||||
|
|
||||||
!!! warning "Ill-formed UTF-8 in text strings"
|
!!! warning "UTF-8 validation of text strings"
|
||||||
|
|
||||||
[RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings
|
[RFC 8949, Section 3.1](https://www.rfc-editor.org/rfc/rfc8949.html#section-3.1) requires CBOR text strings
|
||||||
(major type 3) to be valid UTF-8, but leaves it up to the decoder whether to enforce this. This library does
|
(major type 3) to be valid UTF-8. This library validates the bytes of every text string (object keys included) at
|
||||||
not: `from_cbor()` accepts a text string (object keys included) whose bytes are not valid UTF-8 and hands them
|
decode time and rejects ill-formed UTF-8 with a
|
||||||
back unchanged. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws
|
[`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or, with
|
||||||
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error
|
`allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting value is
|
||||||
handler is passed that replaces or ignores the ill-formed bytes. `to_cbor()` is strict as well and throws the
|
dumped. Byte strings (major type 2) are unaffected and are never validated, since they are not required to hold
|
||||||
same exception for a string value or object key that is not valid UTF-8, so such a value cannot be written back
|
text.
|
||||||
to CBOR. Byte strings (major type 2) are unaffected, since they are not required to hold text.
|
|
||||||
|
|
||||||
!!! warning "Tagged items"
|
!!! warning "Tagged items"
|
||||||
|
|
||||||
|
|||||||
@@ -153,15 +153,14 @@ The library maps MessagePack types to JSON value types as follows:
|
|||||||
This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed
|
This applies to the [SAX interface](../parsing/sax_interface.md) as well, as the key is read before it is passed
|
||||||
on. Such input needs a general-purpose MessagePack library instead.
|
on. Such input needs a general-purpose MessagePack library instead.
|
||||||
|
|
||||||
!!! warning "Ill-formed UTF-8 in string values"
|
!!! warning "UTF-8 validation of string values"
|
||||||
|
|
||||||
The MessagePack specification explicitly allows a `str` value (`fixstr`, `str 8`, `str 16`, `str 32`) to contain
|
The MessagePack specification requires `str` values (`fixstr`, `str 8`, `str 16`, `str 32`) to be valid UTF-8.
|
||||||
a byte sequence that is not valid UTF-8, and expects a deserializer to hand the original bytes back unchanged.
|
This library validates the bytes of every such string (object keys included) at decode time and rejects
|
||||||
This library follows that: `from_msgpack()` reads `str` bytes (object keys included) as-is, without validating
|
ill-formed UTF-8 with a [`parse_error.113`](../../home/exceptions.md#jsonexceptionparse_error113) exception (or,
|
||||||
them, and `to_msgpack()` writes them back as-is, so such a value round-trips through `from_msgpack(to_msgpack(j))`
|
with `allow_exceptions` set to `false`, a discarded value), rather than only failing later when the resulting
|
||||||
byte for byte. However, [`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws
|
value is dumped. `bin`/`ext`/`fixext` values are unaffected and are never validated, since they are not required
|
||||||
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for a value read this way, unless an
|
to hold text.
|
||||||
error handler is passed that replaces or ignores the ill-formed bytes.
|
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
|
|
||||||
|
|||||||
@@ -47,12 +47,6 @@ The library uses the following mapping from JSON values types to UBJSON types ac
|
|||||||
|
|
||||||
- strings with more than 9223372036854775807 bytes (theoretical)
|
- strings with more than 9223372036854775807 bytes (theoretical)
|
||||||
|
|
||||||
!!! warning "UTF-8 validation of string values and object keys"
|
|
||||||
|
|
||||||
UBJSON's required string encoding is UTF-8. `to_ubjson()` validates the bytes of every string value and object
|
|
||||||
key and throws [`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for ill-formed UTF-8, so
|
|
||||||
a value with such a string cannot be serialized in the first place.
|
|
||||||
|
|
||||||
!!! info "Unused UBJSON markers"
|
!!! info "Unused UBJSON markers"
|
||||||
|
|
||||||
The following markers are not used in the conversion:
|
The following markers are not used in the conversion:
|
||||||
@@ -126,15 +120,6 @@ The library maps UBJSON types to JSON value types as follows:
|
|||||||
|
|
||||||
The mapping is **complete** in the sense that any UBJSON value can be converted to a JSON value.
|
The mapping is **complete** in the sense that any UBJSON value can be converted to a JSON value.
|
||||||
|
|
||||||
!!! warning "Ill-formed UTF-8 in string values and object keys"
|
|
||||||
|
|
||||||
UBJSON's required string encoding is UTF-8, but this is not enforced on read: `from_ubjson()` accepts a string
|
|
||||||
value or object key whose bytes are not valid UTF-8 and hands them back unchanged. However,
|
|
||||||
[`dump()`](../../api/basic_json/dump.md) still requires valid UTF-8 and throws
|
|
||||||
[`type_error.316`](../../home/exceptions.md#jsonexceptiontype_error316) for such a value, unless an error
|
|
||||||
handler is passed that replaces or ignores the ill-formed bytes. `to_ubjson()` is strict as well (see above), so
|
|
||||||
a value read this way cannot be written back to UBJSON.
|
|
||||||
|
|
||||||
??? example
|
??? example
|
||||||
|
|
||||||
```cpp
|
```cpp
|
||||||
|
|||||||
@@ -340,9 +340,8 @@ An unexpected byte was read in a [binary format](../features/binary_formats/inde
|
|||||||
### json.exception.parse_error.113
|
### json.exception.parse_error.113
|
||||||
|
|
||||||
A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a
|
A string could not be read from a [binary format](../features/binary_formats/index.md): either a value that is not a
|
||||||
string was read where one was required (for instance as a map key), or the string's length specification is invalid.
|
string was read where one was required (for instance as a map key), the string's length specification is invalid, or
|
||||||
The bytes of a string itself are not checked for valid UTF-8 on read; see the ill-formed UTF-8 notes on the
|
the string's bytes are not valid UTF-8.
|
||||||
individual [binary format](../features/binary_formats/index.md) pages for how such a string is handled afterward.
|
|
||||||
|
|
||||||
CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other
|
CBOR and MessagePack allow map keys of any type, but JSON object keys are always strings. Maps with keys of any other
|
||||||
type (for instance integers or `null`) are therefore not supported; see the notes on
|
type (for instance integers or `null`) are therefore not supported; see the notes on
|
||||||
@@ -365,6 +364,9 @@ type (for instance integers or `null`) are therefore not supported; see the note
|
|||||||
```
|
```
|
||||||
[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative
|
[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing BJData string: string length must not be negative
|
||||||
```
|
```
|
||||||
|
```
|
||||||
|
[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte
|
||||||
|
```
|
||||||
|
|
||||||
### json.exception.parse_error.114
|
### json.exception.parse_error.114
|
||||||
|
|
||||||
|
|||||||
@@ -4031,13 +4031,28 @@ class binary_reader
|
|||||||
const NumberType len,
|
const NumberType len,
|
||||||
string_t& result)
|
string_t& result)
|
||||||
{
|
{
|
||||||
// Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the
|
// get_bytes() appends to result, and CBOR indefinite-length strings
|
||||||
// choice to the decoder), MessagePack (whose spec explicitly allows
|
// collect all their chunks in the same result; validating only the
|
||||||
// a str object to contain an invalid byte sequence), UBJSON, BJData,
|
// newly read bytes keeps the check linear in the input size
|
||||||
// or BSON requires a decoder to reject ill-formed UTF-8. The bytes
|
const std::size_t old_size = result.size();
|
||||||
// are kept unchanged; dump() and the binary writers are the ones
|
if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result)))
|
||||||
// that check them and report type_error.316 if they are not valid.
|
{
|
||||||
return get_bytes(format, len, "string", result);
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications
|
||||||
|
// all require text strings to be valid UTF-8; reject anything else
|
||||||
|
// right here so malformed input is caught at decode time instead of
|
||||||
|
// only surfacing later as a type_error.316 when the value is dumped
|
||||||
|
// (which would defeat allow_exceptions=false / strict discarding).
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size)))
|
||||||
|
{
|
||||||
|
return sax->parse_error(chars_read, get_token_string(),
|
||||||
|
parse_error::create(113, chars_read,
|
||||||
|
exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr));
|
||||||
|
}
|
||||||
|
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
|
|||||||
@@ -115,8 +115,6 @@ class binary_writer
|
|||||||
|
|
||||||
/*!
|
/*!
|
||||||
@param[in] j JSON value to serialize
|
@param[in] j JSON value to serialize
|
||||||
@throw type_error.316 if a string value or an object key is not valid
|
|
||||||
UTF-8
|
|
||||||
@throw type_error.317 if @a j is not an object
|
@throw type_error.317 if @a j is not an object
|
||||||
*/
|
*/
|
||||||
void write_bson(const BasicJsonType& j)
|
void write_bson(const BasicJsonType& j)
|
||||||
@@ -147,8 +145,6 @@ class binary_writer
|
|||||||
|
|
||||||
/*!
|
/*!
|
||||||
@param[in] j JSON value to serialize
|
@param[in] j JSON value to serialize
|
||||||
@throw type_error.316 if a string value or an object key is not valid
|
|
||||||
UTF-8
|
|
||||||
*/
|
*/
|
||||||
void write_cbor(const BasicJsonType& j)
|
void write_cbor(const BasicJsonType& j)
|
||||||
{
|
{
|
||||||
@@ -215,8 +211,6 @@ class binary_writer
|
|||||||
|
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
{
|
{
|
||||||
check_utf8(*j.m_data.m_value.string, j);
|
|
||||||
|
|
||||||
// step 1: write control byte and the string length
|
// step 1: write control byte and the string length
|
||||||
write_cbor_head(0x60, j.m_data.m_value.string->size());
|
write_cbor_head(0x60, j.m_data.m_value.string->size());
|
||||||
|
|
||||||
@@ -293,11 +287,6 @@ class binary_writer
|
|||||||
// step 2: write each element
|
// step 2: write each element
|
||||||
for (const auto& el : *j.m_data.m_value.object)
|
for (const auto& el : *j.m_data.m_value.object)
|
||||||
{
|
{
|
||||||
// el.first is checked here, against the object as
|
|
||||||
// diagnostics context, because write_cbor(el.first)
|
|
||||||
// converts it to a temporary basic_json that would be
|
|
||||||
// used as the context instead
|
|
||||||
check_utf8(el.first, j);
|
|
||||||
write_cbor(el.first);
|
write_cbor(el.first);
|
||||||
write_cbor(el.second);
|
write_cbor(el.second);
|
||||||
}
|
}
|
||||||
@@ -640,8 +629,6 @@ class binary_writer
|
|||||||
@param[in] add_prefix whether prefixes need to be used for this value
|
@param[in] add_prefix whether prefixes need to be used for this value
|
||||||
@param[in] use_bjdata whether write in BJData format, default is false
|
@param[in] use_bjdata whether write in BJData format, default is false
|
||||||
@param[in] bjdata_version which BJData version to use, default is draft2
|
@param[in] bjdata_version which BJData version to use, default is draft2
|
||||||
@throw type_error.316 if a string value or an object key is not valid
|
|
||||||
UTF-8
|
|
||||||
*/
|
*/
|
||||||
void write_ubjson(const BasicJsonType& j, const bool use_count,
|
void write_ubjson(const BasicJsonType& j, const bool use_count,
|
||||||
const bool use_type, const bool add_prefix = true,
|
const bool use_type, const bool add_prefix = true,
|
||||||
@@ -691,8 +678,6 @@ class binary_writer
|
|||||||
|
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
{
|
{
|
||||||
check_utf8(*j.m_data.m_value.string, j);
|
|
||||||
|
|
||||||
if (add_prefix)
|
if (add_prefix)
|
||||||
{
|
{
|
||||||
oa.write_character(to_char_type('S'));
|
oa.write_character(to_char_type('S'));
|
||||||
@@ -855,7 +840,6 @@ class binary_writer
|
|||||||
|
|
||||||
for (const auto& el : *j.m_data.m_value.object)
|
for (const auto& el : *j.m_data.m_value.object)
|
||||||
{
|
{
|
||||||
check_utf8(el.first, j);
|
|
||||||
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
|
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
|
||||||
oa.write_characters(
|
oa.write_characters(
|
||||||
reinterpret_cast<const CharType*>(el.first.data()),
|
reinterpret_cast<const CharType*>(el.first.data()),
|
||||||
@@ -900,10 +884,6 @@ class binary_writer
|
|||||||
/*!
|
/*!
|
||||||
@return The size of a BSON document entry header, including the id marker
|
@return The size of a BSON document entry header, including the id marker
|
||||||
and the entry name size (and its null-terminator).
|
and the entry name size (and its null-terminator).
|
||||||
@throw out_of_range.409 if @a name contains U+0000, before anything is
|
|
||||||
written
|
|
||||||
@throw type_error.316 if @a name is not valid UTF-8, before anything is
|
|
||||||
written
|
|
||||||
*/
|
*/
|
||||||
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
||||||
{
|
{
|
||||||
@@ -913,8 +893,7 @@ class binary_writer
|
|||||||
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
|
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
|
||||||
}
|
}
|
||||||
|
|
||||||
check_utf8(name, j);
|
static_cast<void>(j);
|
||||||
|
|
||||||
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
|
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -970,21 +949,9 @@ class binary_writer
|
|||||||
|
|
||||||
/*!
|
/*!
|
||||||
@return The size of the BSON-encoded string in @a value
|
@return The size of the BSON-encoded string in @a value
|
||||||
@throw type_error.316 if @a value is not valid UTF-8, before anything is
|
|
||||||
written
|
|
||||||
|
|
||||||
@note The UTF-8 check is skipped if @a value is already too long for the
|
|
||||||
32-bit BSON length field (@ref to_bson_length rejects it later, once
|
|
||||||
the size of the whole document is known); this also keeps the check
|
|
||||||
from reading past a StringType that reports a size larger than what
|
|
||||||
it actually holds.
|
|
||||||
*/
|
*/
|
||||||
static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
|
static std::size_t calc_bson_string_size(const string_t& value)
|
||||||
{
|
{
|
||||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<std::int32_t>(value.size())))
|
|
||||||
{
|
|
||||||
check_utf8(value, j);
|
|
||||||
}
|
|
||||||
return sizeof(std::int32_t) + value.size() + 1ul;
|
return sizeof(std::int32_t) + value.size() + 1ul;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1113,8 +1080,6 @@ class binary_writer
|
|||||||
is neither an object nor an array
|
is neither an object nor an array
|
||||||
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
|
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
|
||||||
into a byte, before anything is written
|
into a byte, before anything is written
|
||||||
@throw type_error.316 if @a j is a string that is not valid UTF-8, before
|
|
||||||
anything is written
|
|
||||||
*/
|
*/
|
||||||
static std::size_t calc_bson_value_size(const BasicJsonType& j)
|
static std::size_t calc_bson_value_size(const BasicJsonType& j)
|
||||||
{
|
{
|
||||||
@@ -1136,7 +1101,7 @@ class binary_writer
|
|||||||
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
|
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
|
||||||
|
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
return calc_bson_string_size(*j.m_data.m_value.string, j);
|
return calc_bson_string_size(*j.m_data.m_value.string);
|
||||||
|
|
||||||
case value_t::null:
|
case value_t::null:
|
||||||
return 0ul;
|
return 0ul;
|
||||||
@@ -1249,8 +1214,6 @@ class binary_writer
|
|||||||
written
|
written
|
||||||
@throw out_of_range.415 if a binary value's subtype does not fit into a
|
@throw out_of_range.415 if a binary value's subtype does not fit into a
|
||||||
byte, before anything is written
|
byte, before anything is written
|
||||||
@throw type_error.316 if a string value or a key is not valid UTF-8,
|
|
||||||
before anything is written
|
|
||||||
*/
|
*/
|
||||||
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
||||||
{
|
{
|
||||||
@@ -2129,7 +2092,7 @@ class binary_writer
|
|||||||
*/
|
*/
|
||||||
void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context)
|
void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context)
|
||||||
{
|
{
|
||||||
check_utf8(s, context);
|
check_bon8_utf8(s, context);
|
||||||
|
|
||||||
// a string that follows another string terminates it
|
// a string that follows another string terminates it
|
||||||
if (string_open)
|
if (string_open)
|
||||||
@@ -2159,7 +2122,7 @@ class binary_writer
|
|||||||
@throw type_error.316 if @a s is not valid UTF-8; the message names the
|
@throw type_error.316 if @a s is not valid UTF-8; the message names the
|
||||||
first byte of the first invalid or incomplete sequence
|
first byte of the first invalid or incomplete sequence
|
||||||
*/
|
*/
|
||||||
static void check_utf8(const string_t& s, const BasicJsonType& context)
|
static void check_bon8_utf8(const string_t& s, const BasicJsonType& context)
|
||||||
{
|
{
|
||||||
static_cast<void>(context); // only used when exceptions are enabled
|
static_cast<void>(context); // only used when exceptions are enabled
|
||||||
const auto* data = reinterpret_cast<const unsigned char*>(s.data());
|
const auto* data = reinterpret_cast<const unsigned char*>(s.data());
|
||||||
|
|||||||
@@ -117,14 +117,13 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
|
|||||||
written by Björn Hoehrmann. See
|
written by Björn Hoehrmann. See
|
||||||
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
|
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
|
||||||
|
|
||||||
The library checks UTF-8 well-formedness (RFC 3629, section 4) in three
|
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
|
||||||
places, which differ in speed, diagnostics, and how they read the input:
|
places, which differ in speed, diagnostics, and how they read the input:
|
||||||
|
|
||||||
- decode() below: the serializer, to escape and, in strict mode, reject
|
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
|
||||||
ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON,
|
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
|
||||||
UBJSON and BJData readers do not use it: none of those specs requires a
|
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
|
||||||
decoder to reject ill-formed UTF-8 in text strings, so the readers keep
|
text strings at decode time).
|
||||||
the bytes as is and leave the check to dump() and the binary writers.
|
|
||||||
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
|
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
|
||||||
diagnostic for each kind of error.
|
diagnostic for each kind of error.
|
||||||
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
|
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
|
||||||
@@ -179,5 +178,38 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std:
|
|||||||
return state;
|
return state;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief check whether a string consists solely of valid UTF-8
|
||||||
|
|
||||||
|
Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text
|
||||||
|
strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the
|
||||||
|
MessagePack/BSON specifications all require text strings to be UTF-8), so
|
||||||
|
that malformed input is caught immediately instead of only surfacing later
|
||||||
|
as a type_error.316 when the resulting value is dumped.
|
||||||
|
|
||||||
|
@param[in] s the string to check
|
||||||
|
@param[in] first index of the first byte to check; the bytes before it are
|
||||||
|
assumed to have been validated already and to end on a
|
||||||
|
code point boundary
|
||||||
|
@return whether @a s (from index @a first on) is valid UTF-8
|
||||||
|
*/
|
||||||
|
template<typename StringType>
|
||||||
|
inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept
|
||||||
|
{
|
||||||
|
std::uint8_t state = UTF8_ACCEPT;
|
||||||
|
std::uint32_t codepoint = 0;
|
||||||
|
|
||||||
|
for (std::size_t i = first; i < s.size(); ++i)
|
||||||
|
{
|
||||||
|
decode(state, codepoint, static_cast<std::uint8_t>(s[i]));
|
||||||
|
if (state == UTF8_REJECT)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return state == UTF8_ACCEPT;
|
||||||
|
}
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
NLOHMANN_JSON_NAMESPACE_END
|
NLOHMANN_JSON_NAMESPACE_END
|
||||||
|
|||||||
@@ -6321,14 +6321,13 @@ This is a single-byte step of a "shift-based" UTF-8 decoder originally
|
|||||||
written by Björn Hoehrmann. See
|
written by Björn Hoehrmann. See
|
||||||
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
|
http://bjoern.hoehrmann.de/utf-8/decoder/dfa/ for details.
|
||||||
|
|
||||||
The library checks UTF-8 well-formedness (RFC 3629, section 4) in three
|
The library checks UTF-8 well-formedness (RFC 3629, section 4) in four
|
||||||
places, which differ in speed, diagnostics, and how they read the input:
|
places, which differ in speed, diagnostics, and how they read the input:
|
||||||
|
|
||||||
- decode() below: the serializer, to escape and, in strict mode, reject
|
- decode() and @ref is_valid_utf8 below: the serializer (to escape and, in
|
||||||
ill-formed UTF-8 when dumping a string. The CBOR, MessagePack, BSON,
|
strict mode, reject ill-formed UTF-8 when dumping a string) and the CBOR,
|
||||||
UBJSON and BJData readers do not use it: none of those specs requires a
|
MessagePack, BSON, UBJSON and BJData readers (to reject ill-formed UTF-8 in
|
||||||
decoder to reject ill-formed UTF-8 in text strings, so the readers keep
|
text strings at decode time).
|
||||||
the bytes as is and leave the check to dump() and the binary writers.
|
|
||||||
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
|
- the per-lead-byte switch in lexer::scan_string(): JSON text, with a
|
||||||
diagnostic for each kind of error.
|
diagnostic for each kind of error.
|
||||||
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
|
- validate_one_utf8() and valid_utf8_prefix() in string_scan.hpp: the lexer's
|
||||||
@@ -6383,6 +6382,39 @@ inline std::uint8_t decode(std::uint8_t& state, std::uint32_t& codep, const std:
|
|||||||
return state;
|
return state;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*!
|
||||||
|
@brief check whether a string consists solely of valid UTF-8
|
||||||
|
|
||||||
|
Used by the CBOR/MessagePack/BSON/UBJSON binary readers to reject text
|
||||||
|
strings that are not valid UTF-8 at decode time (RFC 8949 §3.1 and the
|
||||||
|
MessagePack/BSON specifications all require text strings to be UTF-8), so
|
||||||
|
that malformed input is caught immediately instead of only surfacing later
|
||||||
|
as a type_error.316 when the resulting value is dumped.
|
||||||
|
|
||||||
|
@param[in] s the string to check
|
||||||
|
@param[in] first index of the first byte to check; the bytes before it are
|
||||||
|
assumed to have been validated already and to end on a
|
||||||
|
code point boundary
|
||||||
|
@return whether @a s (from index @a first on) is valid UTF-8
|
||||||
|
*/
|
||||||
|
template<typename StringType>
|
||||||
|
inline bool is_valid_utf8(const StringType& s, const std::size_t first = 0) noexcept
|
||||||
|
{
|
||||||
|
std::uint8_t state = UTF8_ACCEPT;
|
||||||
|
std::uint32_t codepoint = 0;
|
||||||
|
|
||||||
|
for (std::size_t i = first; i < s.size(); ++i)
|
||||||
|
{
|
||||||
|
decode(state, codepoint, static_cast<std::uint8_t>(s[i]));
|
||||||
|
if (state == UTF8_REJECT)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return state == UTF8_ACCEPT;
|
||||||
|
}
|
||||||
|
|
||||||
} // namespace detail
|
} // namespace detail
|
||||||
NLOHMANN_JSON_NAMESPACE_END
|
NLOHMANN_JSON_NAMESPACE_END
|
||||||
|
|
||||||
@@ -17512,13 +17544,28 @@ class binary_reader
|
|||||||
const NumberType len,
|
const NumberType len,
|
||||||
string_t& result)
|
string_t& result)
|
||||||
{
|
{
|
||||||
// Strings are taken as is: none of CBOR (RFC 8949 §3.1 leaves the
|
// get_bytes() appends to result, and CBOR indefinite-length strings
|
||||||
// choice to the decoder), MessagePack (whose spec explicitly allows
|
// collect all their chunks in the same result; validating only the
|
||||||
// a str object to contain an invalid byte sequence), UBJSON, BJData,
|
// newly read bytes keeps the check linear in the input size
|
||||||
// or BSON requires a decoder to reject ill-formed UTF-8. The bytes
|
const std::size_t old_size = result.size();
|
||||||
// are kept unchanged; dump() and the binary writers are the ones
|
if (JSON_HEDLEY_UNLIKELY(!get_bytes(format, len, "string", result)))
|
||||||
// that check them and report type_error.316 if they are not valid.
|
{
|
||||||
return get_bytes(format, len, "string", result);
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// RFC 8949 (CBOR) §3.1 and the MessagePack/BSON/UBJSON specifications
|
||||||
|
// all require text strings to be valid UTF-8; reject anything else
|
||||||
|
// right here so malformed input is caught at decode time instead of
|
||||||
|
// only surfacing later as a type_error.316 when the value is dumped
|
||||||
|
// (which would defeat allow_exceptions=false / strict discarding).
|
||||||
|
if (JSON_HEDLEY_UNLIKELY(!is_valid_utf8(result, old_size)))
|
||||||
|
{
|
||||||
|
return sax->parse_error(chars_read, get_token_string(),
|
||||||
|
parse_error::create(113, chars_read,
|
||||||
|
exception_message(format, "invalid string: ill-formed UTF-8 byte", "string"), nullptr));
|
||||||
|
}
|
||||||
|
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
/*!
|
/*!
|
||||||
@@ -21193,8 +21240,6 @@ class binary_writer
|
|||||||
|
|
||||||
/*!
|
/*!
|
||||||
@param[in] j JSON value to serialize
|
@param[in] j JSON value to serialize
|
||||||
@throw type_error.316 if a string value or an object key is not valid
|
|
||||||
UTF-8
|
|
||||||
@throw type_error.317 if @a j is not an object
|
@throw type_error.317 if @a j is not an object
|
||||||
*/
|
*/
|
||||||
void write_bson(const BasicJsonType& j)
|
void write_bson(const BasicJsonType& j)
|
||||||
@@ -21225,8 +21270,6 @@ class binary_writer
|
|||||||
|
|
||||||
/*!
|
/*!
|
||||||
@param[in] j JSON value to serialize
|
@param[in] j JSON value to serialize
|
||||||
@throw type_error.316 if a string value or an object key is not valid
|
|
||||||
UTF-8
|
|
||||||
*/
|
*/
|
||||||
void write_cbor(const BasicJsonType& j)
|
void write_cbor(const BasicJsonType& j)
|
||||||
{
|
{
|
||||||
@@ -21293,8 +21336,6 @@ class binary_writer
|
|||||||
|
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
{
|
{
|
||||||
check_utf8(*j.m_data.m_value.string, j);
|
|
||||||
|
|
||||||
// step 1: write control byte and the string length
|
// step 1: write control byte and the string length
|
||||||
write_cbor_head(0x60, j.m_data.m_value.string->size());
|
write_cbor_head(0x60, j.m_data.m_value.string->size());
|
||||||
|
|
||||||
@@ -21371,11 +21412,6 @@ class binary_writer
|
|||||||
// step 2: write each element
|
// step 2: write each element
|
||||||
for (const auto& el : *j.m_data.m_value.object)
|
for (const auto& el : *j.m_data.m_value.object)
|
||||||
{
|
{
|
||||||
// el.first is checked here, against the object as
|
|
||||||
// diagnostics context, because write_cbor(el.first)
|
|
||||||
// converts it to a temporary basic_json that would be
|
|
||||||
// used as the context instead
|
|
||||||
check_utf8(el.first, j);
|
|
||||||
write_cbor(el.first);
|
write_cbor(el.first);
|
||||||
write_cbor(el.second);
|
write_cbor(el.second);
|
||||||
}
|
}
|
||||||
@@ -21718,8 +21754,6 @@ class binary_writer
|
|||||||
@param[in] add_prefix whether prefixes need to be used for this value
|
@param[in] add_prefix whether prefixes need to be used for this value
|
||||||
@param[in] use_bjdata whether write in BJData format, default is false
|
@param[in] use_bjdata whether write in BJData format, default is false
|
||||||
@param[in] bjdata_version which BJData version to use, default is draft2
|
@param[in] bjdata_version which BJData version to use, default is draft2
|
||||||
@throw type_error.316 if a string value or an object key is not valid
|
|
||||||
UTF-8
|
|
||||||
*/
|
*/
|
||||||
void write_ubjson(const BasicJsonType& j, const bool use_count,
|
void write_ubjson(const BasicJsonType& j, const bool use_count,
|
||||||
const bool use_type, const bool add_prefix = true,
|
const bool use_type, const bool add_prefix = true,
|
||||||
@@ -21769,8 +21803,6 @@ class binary_writer
|
|||||||
|
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
{
|
{
|
||||||
check_utf8(*j.m_data.m_value.string, j);
|
|
||||||
|
|
||||||
if (add_prefix)
|
if (add_prefix)
|
||||||
{
|
{
|
||||||
oa.write_character(to_char_type('S'));
|
oa.write_character(to_char_type('S'));
|
||||||
@@ -21933,7 +21965,6 @@ class binary_writer
|
|||||||
|
|
||||||
for (const auto& el : *j.m_data.m_value.object)
|
for (const auto& el : *j.m_data.m_value.object)
|
||||||
{
|
{
|
||||||
check_utf8(el.first, j);
|
|
||||||
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
|
write_number_with_ubjson_prefix(el.first.size(), true, use_bjdata);
|
||||||
oa.write_characters(
|
oa.write_characters(
|
||||||
reinterpret_cast<const CharType*>(el.first.data()),
|
reinterpret_cast<const CharType*>(el.first.data()),
|
||||||
@@ -21978,10 +22009,6 @@ class binary_writer
|
|||||||
/*!
|
/*!
|
||||||
@return The size of a BSON document entry header, including the id marker
|
@return The size of a BSON document entry header, including the id marker
|
||||||
and the entry name size (and its null-terminator).
|
and the entry name size (and its null-terminator).
|
||||||
@throw out_of_range.409 if @a name contains U+0000, before anything is
|
|
||||||
written
|
|
||||||
@throw type_error.316 if @a name is not valid UTF-8, before anything is
|
|
||||||
written
|
|
||||||
*/
|
*/
|
||||||
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
static std::size_t calc_bson_entry_header_size(const string_t& name, const BasicJsonType& j)
|
||||||
{
|
{
|
||||||
@@ -21991,8 +22018,7 @@ class binary_writer
|
|||||||
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
|
JSON_THROW(out_of_range::create(409, concat("BSON key cannot contain code point U+0000 (at byte ", std::to_string(it), ")"), &j));
|
||||||
}
|
}
|
||||||
|
|
||||||
check_utf8(name, j);
|
static_cast<void>(j);
|
||||||
|
|
||||||
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
|
return /*id*/ 1ul + name.size() + /*zero-terminator*/1u;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -22048,21 +22074,9 @@ class binary_writer
|
|||||||
|
|
||||||
/*!
|
/*!
|
||||||
@return The size of the BSON-encoded string in @a value
|
@return The size of the BSON-encoded string in @a value
|
||||||
@throw type_error.316 if @a value is not valid UTF-8, before anything is
|
|
||||||
written
|
|
||||||
|
|
||||||
@note The UTF-8 check is skipped if @a value is already too long for the
|
|
||||||
32-bit BSON length field (@ref to_bson_length rejects it later, once
|
|
||||||
the size of the whole document is known); this also keeps the check
|
|
||||||
from reading past a StringType that reports a size larger than what
|
|
||||||
it actually holds.
|
|
||||||
*/
|
*/
|
||||||
static std::size_t calc_bson_string_size(const string_t& value, const BasicJsonType& j)
|
static std::size_t calc_bson_string_size(const string_t& value)
|
||||||
{
|
{
|
||||||
if (JSON_HEDLEY_LIKELY(value_in_range_of<std::int32_t>(value.size())))
|
|
||||||
{
|
|
||||||
check_utf8(value, j);
|
|
||||||
}
|
|
||||||
return sizeof(std::int32_t) + value.size() + 1ul;
|
return sizeof(std::int32_t) + value.size() + 1ul;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -22191,8 +22205,6 @@ class binary_writer
|
|||||||
is neither an object nor an array
|
is neither an object nor an array
|
||||||
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
|
@throw out_of_range.415 if @a j is binary with a subtype that does not fit
|
||||||
into a byte, before anything is written
|
into a byte, before anything is written
|
||||||
@throw type_error.316 if @a j is a string that is not valid UTF-8, before
|
|
||||||
anything is written
|
|
||||||
*/
|
*/
|
||||||
static std::size_t calc_bson_value_size(const BasicJsonType& j)
|
static std::size_t calc_bson_value_size(const BasicJsonType& j)
|
||||||
{
|
{
|
||||||
@@ -22214,7 +22226,7 @@ class binary_writer
|
|||||||
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
|
return calc_bson_unsigned_size(j.m_data.m_value.number_unsigned);
|
||||||
|
|
||||||
case value_t::string:
|
case value_t::string:
|
||||||
return calc_bson_string_size(*j.m_data.m_value.string, j);
|
return calc_bson_string_size(*j.m_data.m_value.string);
|
||||||
|
|
||||||
case value_t::null:
|
case value_t::null:
|
||||||
return 0ul;
|
return 0ul;
|
||||||
@@ -22327,8 +22339,6 @@ class binary_writer
|
|||||||
written
|
written
|
||||||
@throw out_of_range.415 if a binary value's subtype does not fit into a
|
@throw out_of_range.415 if a binary value's subtype does not fit into a
|
||||||
byte, before anything is written
|
byte, before anything is written
|
||||||
@throw type_error.316 if a string value or a key is not valid UTF-8,
|
|
||||||
before anything is written
|
|
||||||
*/
|
*/
|
||||||
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
static std::size_t calc_bson_sizes(const BasicJsonType& document, std::vector<std::size_t>& nested_sizes)
|
||||||
{
|
{
|
||||||
@@ -23207,7 +23217,7 @@ class binary_writer
|
|||||||
*/
|
*/
|
||||||
void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context)
|
void write_bon8_string(const string_t& s, bool& string_open, const BasicJsonType& context)
|
||||||
{
|
{
|
||||||
check_utf8(s, context);
|
check_bon8_utf8(s, context);
|
||||||
|
|
||||||
// a string that follows another string terminates it
|
// a string that follows another string terminates it
|
||||||
if (string_open)
|
if (string_open)
|
||||||
@@ -23237,7 +23247,7 @@ class binary_writer
|
|||||||
@throw type_error.316 if @a s is not valid UTF-8; the message names the
|
@throw type_error.316 if @a s is not valid UTF-8; the message names the
|
||||||
first byte of the first invalid or incomplete sequence
|
first byte of the first invalid or incomplete sequence
|
||||||
*/
|
*/
|
||||||
static void check_utf8(const string_t& s, const BasicJsonType& context)
|
static void check_bon8_utf8(const string_t& s, const BasicJsonType& context)
|
||||||
{
|
{
|
||||||
static_cast<void>(context); // only used when exceptions are enabled
|
static_cast<void>(context); // only used when exceptions are enabled
|
||||||
const auto* data = reinterpret_cast<const unsigned char*>(s.data());
|
const auto* data = reinterpret_cast<const unsigned char*>(s.data());
|
||||||
|
|||||||
@@ -3907,41 +3907,6 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
|||||||
CHECK(json::to_bjdata(j) == v);
|
CHECK(json::to_bjdata(j) == v);
|
||||||
CHECK(json::from_bjdata(v) == j);
|
CHECK(json::from_bjdata(v) == j);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
|
||||||
{
|
|
||||||
// none of the binary format specs requires a decoder to reject
|
|
||||||
// ill-formed UTF-8 in a text string, so a value whose bytes are
|
|
||||||
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
|
|
||||||
// round-trips byte for byte as a string value; to_bjdata() is
|
|
||||||
// strict, so such a value cannot be written back
|
|
||||||
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
|
|
||||||
json j;
|
|
||||||
CHECK_NOTHROW(j = json::from_bjdata(v));
|
|
||||||
REQUIRE(j.is_string());
|
|
||||||
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
|
||||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
|
||||||
CHECK_THROWS_AS(json::to_bjdata(j), json::type_error&);
|
|
||||||
|
|
||||||
// the same bytes as an object key round-trip as well
|
|
||||||
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
|
|
||||||
json j_key;
|
|
||||||
CHECK_NOTHROW(j_key = json::from_bjdata(v_key));
|
|
||||||
REQUIRE(j_key.is_object());
|
|
||||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
|
||||||
CHECK_THROWS_AS(json::to_bjdata(j_key), json::type_error&);
|
|
||||||
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
|
||||||
// a truncated multi-byte sequence
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
|
||||||
// an encoded surrogate half (U+D800)
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
|
||||||
// an overlong encoding of '.'
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
|
||||||
|
|
||||||
// an object key with ill-formed UTF-8 is rejected the same way
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bjdata(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("Array Type")
|
SECTION("Array Type")
|
||||||
|
|||||||
@@ -154,51 +154,6 @@ TEST_CASE("BSON")
|
|||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
|
||||||
{
|
|
||||||
// a BSON document {"s": "\xC0\xAE"} (0xC0 0xAE is an overlong
|
|
||||||
// encoding of '.'); the BSON spec does not require a decoder to
|
|
||||||
// reject ill-formed UTF-8 in a string value, so the reader hands the
|
|
||||||
// bytes back unchanged
|
|
||||||
const std::vector<uint8_t> v =
|
|
||||||
{
|
|
||||||
0x0F, 0x00, 0x00, 0x00, // document length
|
|
||||||
0x02, 's', 0x00, // type 0x02 (string), key "s"
|
|
||||||
0x03, 0x00, 0x00, 0x00, // string length (including null)
|
|
||||||
0xc0, 0xae, 0x00, // string content and its null terminator
|
|
||||||
0x00 // document terminator
|
|
||||||
};
|
|
||||||
json j;
|
|
||||||
CHECK_NOTHROW(j = json::from_bson(v));
|
|
||||||
REQUIRE(j.is_object());
|
|
||||||
REQUIRE(j.contains("s"));
|
|
||||||
CHECK(j["s"].get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
|
||||||
// dump() still requires valid UTF-8 and throws for such a value
|
|
||||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
|
||||||
// to_bson() is strict as well, so the value cannot be written back
|
|
||||||
CHECK_THROWS_AS(json::to_bson(j), json::type_error&);
|
|
||||||
|
|
||||||
// to_bson() rejects the same kind of ill-formed string value, before
|
|
||||||
// any bytes reach the output adapter (the BSON document length
|
|
||||||
// prefix must be known up front, so nothing is written incrementally)
|
|
||||||
std::vector<std::uint8_t> out{0x42}; // a sentinel byte the writer must not touch
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}, nlohmann::detail::output_adapter<std::uint8_t>(out)), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
|
||||||
CHECK(out == std::vector<std::uint8_t> {0x42});
|
|
||||||
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xFF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
|
||||||
// a truncated multi-byte sequence
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC3"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
|
||||||
// an encoded surrogate half (U+D800)
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xED\xA0\x80"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
|
||||||
// an overlong encoding of '.'
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"s", "\xC0\xAF"}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
|
||||||
|
|
||||||
// an object key with ill-formed UTF-8 is rejected as well; unlike
|
|
||||||
// the reader (which never validates element names), the writer
|
|
||||||
// checks both string values and object keys
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_bson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON")
|
SECTION("lengths exceeding INT32_MAX cannot be serialized to BSON")
|
||||||
{
|
{
|
||||||
// out_of_range.412 is thrown from a single shared helper
|
// out_of_range.412 is thrown from a single shared helper
|
||||||
|
|||||||
+18
-65
@@ -1801,40 +1801,19 @@ TEST_CASE("CBOR")
|
|||||||
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0xA1, 0x7C, 0x01})), "[json.exception.parse_error.113] parse error at byte 2: syntax error while parsing CBOR string: expected length specification (0x60-0x7B) or indefinite string type (0x7F); last byte: 0x7C", json::parse_error&);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("ill-formed UTF-8 in string (see #5529, #5651)")
|
SECTION("invalid UTF-8 in string (see #5529)")
|
||||||
{
|
{
|
||||||
// RFC 8949 §3.1 leaves it up to the decoder whether to reject
|
|
||||||
// ill-formed UTF-8 in a text string; this library does not, and
|
|
||||||
// hands the original bytes back unchanged, matching the
|
|
||||||
// MessagePack reader and the behavior before #5185/#5531 (not in
|
|
||||||
// any release)
|
|
||||||
|
|
||||||
// a two-character text string (major type 3) whose bytes are not
|
// a two-character text string (major type 3) whose bytes are not
|
||||||
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') round-trips
|
// valid UTF-8 (0xC0 0xAE is an overlong encoding of '.') must be
|
||||||
// byte for byte as a string value
|
// rejected at decode time, matching every other kind of
|
||||||
const std::vector<uint8_t> ill_formed_value = {0x62, 0xc0, 0xae};
|
// malformed binary input, rather than only failing later when
|
||||||
json j_value;
|
// the resulting value is dumped
|
||||||
CHECK_NOTHROW(j_value = json::from_cbor(ill_formed_value));
|
json _;
|
||||||
REQUIRE(j_value.is_string());
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||||
CHECK(j_value.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x62, 0xc0, 0xae}), true, false).is_discarded());
|
||||||
// dump() still requires valid UTF-8 and throws for such a value,
|
|
||||||
// unless an error handler that replaces or ignores the bytes is
|
|
||||||
// passed
|
|
||||||
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
|
|
||||||
// to_cbor() is strict as well, so the value cannot be written back
|
|
||||||
CHECK_THROWS_AS(json::to_cbor(j_value), json::type_error&);
|
|
||||||
|
|
||||||
// the same bytes as an object key round-trip as well
|
|
||||||
const std::vector<uint8_t> ill_formed_key = {0xa1, 0x62, 0xc0, 0xae, 0x01};
|
|
||||||
json j_key;
|
|
||||||
CHECK_NOTHROW(j_key = json::from_cbor(ill_formed_key));
|
|
||||||
REQUIRE(j_key.is_object());
|
|
||||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
|
||||||
CHECK_THROWS_AS(json::to_cbor(j_key), json::type_error&);
|
|
||||||
|
|
||||||
// a CBOR byte string (major type 2) with the very same bytes is
|
// a CBOR byte string (major type 2) with the very same bytes is
|
||||||
// NOT text and must still be accepted as-is
|
// NOT text and must still be accepted as-is
|
||||||
json _;
|
|
||||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x42, 0xc0, 0xae})));
|
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x42, 0xc0, 0xae})));
|
||||||
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
||||||
|
|
||||||
@@ -1843,46 +1822,17 @@ TEST_CASE("CBOR")
|
|||||||
CHECK(json::from_cbor(json::to_cbor(j)) == j);
|
CHECK(json::from_cbor(json::to_cbor(j)) == j);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("to_cbor rejects ill-formed UTF-8 (see #5651)")
|
SECTION("invalid UTF-8 in indefinite-length string")
|
||||||
{
|
|
||||||
// to_cbor() must reject the same ill-formed strings from_cbor()
|
|
||||||
// rejects, so a value it accepts can always be read back
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
|
||||||
// a truncated multi-byte sequence
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
|
||||||
// an encoded surrogate half (U+D800)
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
|
||||||
// an overlong encoding of '.'
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_cbor(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
|
||||||
|
|
||||||
// an object key with ill-formed UTF-8 is rejected the same way
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_cbor(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
|
||||||
|
|
||||||
// binary values are not text and are unaffected
|
|
||||||
CHECK_NOTHROW(json::to_cbor(json::binary(std::vector<std::uint8_t>({0xFF}))));
|
|
||||||
}
|
|
||||||
|
|
||||||
SECTION("ill-formed UTF-8 in indefinite-length string")
|
|
||||||
{
|
{
|
||||||
json _;
|
json _;
|
||||||
|
|
||||||
// the chunks are concatenated as is, without checking that each
|
// every chunk must be valid UTF-8 on its own (RFC 8949, Section
|
||||||
// chunk is valid UTF-8 on its own (RFC 8949, Section 3.2.3), so
|
// 3.2.3), so a code point split across two chunks is rejected
|
||||||
// a code point split across two chunks yields a valid string
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff})));
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0x61, 0xa9, 0xff}), true, false).is_discarded());
|
||||||
CHECK(_ == "\xc3\xa9");
|
|
||||||
CHECK(_.dump() == "\"\xc3\xa9\"");
|
|
||||||
|
|
||||||
// a truncated code point is kept as is
|
// an ill-formed later chunk is rejected after valid ones
|
||||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x61, 0xc3, 0xff})));
|
CHECK_THROWS_WITH_AS(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})), "[json.exception.parse_error.113] parse error at byte 7: syntax error while parsing CBOR string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||||
CHECK(_ == "\xc3");
|
|
||||||
CHECK_THROWS_AS(_.dump(), json::type_error&);
|
|
||||||
CHECK_THROWS_AS(json::to_cbor(_), json::type_error&);
|
|
||||||
|
|
||||||
// an ill-formed later chunk is kept after valid ones
|
|
||||||
CHECK_NOTHROW(_ = json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc0, 0xae, 0xff})));
|
|
||||||
CHECK(_ == "\xc3\xa9\xc0\xae");
|
|
||||||
CHECK_THROWS_AS(_.dump(), json::type_error&);
|
|
||||||
|
|
||||||
// valid multi-byte chunks are accepted
|
// valid multi-byte chunks are accepted
|
||||||
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6");
|
CHECK(json::from_cbor(std::vector<uint8_t>({0x7f, 0x62, 0xc3, 0xa9, 0x62, 0xc3, 0xb6, 0xff})) == "\xc3\xa9\xc3\xb6");
|
||||||
@@ -1890,6 +1840,9 @@ TEST_CASE("CBOR")
|
|||||||
|
|
||||||
SECTION("many chunks in indefinite-length string")
|
SECTION("many chunks in indefinite-length string")
|
||||||
{
|
{
|
||||||
|
// only the newly read chunk is validated, not the whole string
|
||||||
|
// collected so far; validating the latter made this input take
|
||||||
|
// quadratic time (about ten seconds for 100000 chunks)
|
||||||
constexpr std::size_t chunks = 100000;
|
constexpr std::size_t chunks = 100000;
|
||||||
std::vector<uint8_t> v{0x7f};
|
std::vector<uint8_t> v{0x7f};
|
||||||
for (std::size_t i = 0; i < chunks; ++i)
|
for (std::size_t i = 0; i < chunks; ++i)
|
||||||
|
|||||||
@@ -1540,39 +1540,19 @@ TEST_CASE("MessagePack")
|
|||||||
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&);
|
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0x81})), "[json.exception.parse_error.110] parse error at byte 2: syntax error while parsing MessagePack string: unexpected end of input", json::parse_error&);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("ill-formed UTF-8 in string (see #5529, #5651)")
|
SECTION("invalid UTF-8 in string (see #5529)")
|
||||||
{
|
{
|
||||||
// the MessagePack specification explicitly allows a str object to
|
|
||||||
// contain a byte sequence that is not valid UTF-8 and expects a
|
|
||||||
// deserializer to hand the original bytes back unchanged; this
|
|
||||||
// library follows that, unlike CBOR/UBJSON/BJData/BSON, whose
|
|
||||||
// specifications require text strings to be valid UTF-8
|
|
||||||
|
|
||||||
// a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8
|
// a fixstr of length 2 (0xA0 | 2) whose bytes are not valid UTF-8
|
||||||
// (0xC0 0xAE is an overlong encoding of '.') round-trips byte for
|
// (0xC0 0xAE is an overlong encoding of '.') must be rejected at
|
||||||
// byte as a string value
|
// decode time, matching every other kind of malformed binary
|
||||||
const std::vector<uint8_t> ill_formed_value = {0xa2, 0xc0, 0xae};
|
// input, rather than only failing later when the resulting
|
||||||
json j_value;
|
// value is dumped
|
||||||
CHECK_NOTHROW(j_value = json::from_msgpack(ill_formed_value));
|
json _;
|
||||||
REQUIRE(j_value.is_string());
|
CHECK_THROWS_WITH_AS(_ = json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae})), "[json.exception.parse_error.113] parse error at byte 3: syntax error while parsing MessagePack string: invalid string: ill-formed UTF-8 byte", json::parse_error&);
|
||||||
CHECK(j_value.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
CHECK(json::from_msgpack(std::vector<uint8_t>({0xa2, 0xc0, 0xae}), true, false).is_discarded());
|
||||||
CHECK(json::from_msgpack(json::to_msgpack(j_value)) == j_value);
|
|
||||||
// dump() still requires valid UTF-8 and throws for such a value,
|
|
||||||
// unless an error handler that replaces or ignores the bytes is
|
|
||||||
// passed
|
|
||||||
CHECK_THROWS_AS(j_value.dump(), json::type_error&);
|
|
||||||
|
|
||||||
// the same bytes as an object key round-trip as well
|
|
||||||
const std::vector<uint8_t> ill_formed_key = {0x81, 0xa2, 0xc0, 0xae, 0x01};
|
|
||||||
json j_key;
|
|
||||||
CHECK_NOTHROW(j_key = json::from_msgpack(ill_formed_key));
|
|
||||||
REQUIRE(j_key.is_object());
|
|
||||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
|
||||||
CHECK(json::from_msgpack(json::to_msgpack(j_key)) == j_key);
|
|
||||||
|
|
||||||
// a MessagePack bin8 blob with the very same bytes is NOT text
|
// a MessagePack bin8 blob with the very same bytes is NOT text
|
||||||
// and must still be accepted as-is
|
// and must still be accepted as-is
|
||||||
json _;
|
|
||||||
CHECK_NOTHROW(_ = json::from_msgpack(std::vector<uint8_t>({0xc4, 0x02, 0xc0, 0xae})));
|
CHECK_NOTHROW(_ = json::from_msgpack(std::vector<uint8_t>({0xc4, 0x02, 0xc0, 0xae})));
|
||||||
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
CHECK(_ == json::binary(std::vector<std::uint8_t>({0xc0, 0xae})));
|
||||||
|
|
||||||
|
|||||||
@@ -2505,41 +2505,6 @@ TEST_CASE("Universal Binary JSON Specification Examples 1")
|
|||||||
CHECK(json::to_ubjson(j) == v);
|
CHECK(json::to_ubjson(j) == v);
|
||||||
CHECK(json::from_ubjson(v) == j);
|
CHECK(json::from_ubjson(v) == j);
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("ill-formed UTF-8 (see #5529, #5651)")
|
|
||||||
{
|
|
||||||
// none of the binary format specs requires a decoder to reject
|
|
||||||
// ill-formed UTF-8 in a text string, so a value whose bytes are
|
|
||||||
// not valid UTF-8 (0xC0 0xAE is an overlong encoding of '.')
|
|
||||||
// round-trips byte for byte as a string value; to_ubjson() is
|
|
||||||
// strict, so such a value cannot be written back
|
|
||||||
const std::vector<uint8_t> v = {'S', 'i', 2, 0xc0, 0xae};
|
|
||||||
json j;
|
|
||||||
CHECK_NOTHROW(j = json::from_ubjson(v));
|
|
||||||
REQUIRE(j.is_string());
|
|
||||||
CHECK(j.get_ref<const json::string_t&>() == std::string("\xc0\xae"));
|
|
||||||
CHECK_THROWS_AS(j.dump(), json::type_error&);
|
|
||||||
CHECK_THROWS_AS(json::to_ubjson(j), json::type_error&);
|
|
||||||
|
|
||||||
// the same bytes as an object key round-trip as well
|
|
||||||
const std::vector<uint8_t> v_key = {'{', 'i', 2, 0xc0, 0xae, 'i', 1, '}'};
|
|
||||||
json j_key;
|
|
||||||
CHECK_NOTHROW(j_key = json::from_ubjson(v_key));
|
|
||||||
REQUIRE(j_key.is_object());
|
|
||||||
CHECK(j_key.contains(std::string("\xc0\xae")));
|
|
||||||
CHECK_THROWS_AS(json::to_ubjson(j_key), json::type_error&);
|
|
||||||
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xFF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
|
||||||
// a truncated multi-byte sequence
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC3")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC3", json::type_error&);
|
|
||||||
// an encoded surrogate half (U+D800)
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xED\xA0\x80")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xED", json::type_error&);
|
|
||||||
// an overlong encoding of '.'
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json("\xC0\xAF")), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xC0", json::type_error&);
|
|
||||||
|
|
||||||
// an object key with ill-formed UTF-8 is rejected the same way
|
|
||||||
CHECK_THROWS_WITH_AS(json::to_ubjson(json{{"\xFF", 1}}), "[json.exception.type_error.316] invalid UTF-8 byte at index 0: 0xFF", json::type_error&);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
SECTION("Array Type")
|
SECTION("Array Type")
|
||||||
|
|||||||
Reference in New Issue
Block a user