mirror of
https://github.com/nlohmann/json.git
synced 2026-09-30 03:30:31 +00:00
Merge remote-tracking branch 'origin/develop' into claude/issue-5340-restore-unget
Signed-off-by: Niels Lohmann <mail@nlohmann.me>
This commit is contained in:
@@ -465,15 +465,6 @@ class binary_reader
|
||||
// CBOR //
|
||||
//////////
|
||||
|
||||
/*!
|
||||
@param[in] get_char whether a new character should be retrieved from the
|
||||
input (true) or whether the last read character should
|
||||
be considered instead (false)
|
||||
@param[in] tag_handler how CBOR tags should be treated
|
||||
|
||||
@return whether a valid CBOR value was passed to the SAX parser
|
||||
*/
|
||||
|
||||
template<typename NumberType>
|
||||
bool get_cbor_negative_integer()
|
||||
{
|
||||
@@ -492,6 +483,14 @@ class binary_reader
|
||||
return sax->number_integer(static_cast<number_integer_t>(-1) - static_cast<number_integer_t>(number));
|
||||
}
|
||||
|
||||
/*!
|
||||
@param[in] get_char whether a new character should be retrieved from the
|
||||
input (true) or whether the last read character should
|
||||
be considered instead (false)
|
||||
@param[in] tag_handler how CBOR tags should be treated
|
||||
|
||||
@return whether a valid CBOR value was passed to the SAX parser
|
||||
*/
|
||||
bool parse_cbor_internal(const bool get_char,
|
||||
const cbor_tag_handler_t tag_handler)
|
||||
{
|
||||
@@ -774,7 +773,13 @@ class binary_reader
|
||||
case 0xBF: // map (indefinite length)
|
||||
return get_cbor_object(detail::unknown_size(), tag_handler);
|
||||
|
||||
case 0xC6: // tagged item
|
||||
case 0xC0: // tagged item
|
||||
case 0xC1:
|
||||
case 0xC2:
|
||||
case 0xC3:
|
||||
case 0xC4:
|
||||
case 0xC5:
|
||||
case 0xC6:
|
||||
case 0xC7:
|
||||
case 0xC8:
|
||||
case 0xC9:
|
||||
@@ -789,6 +794,9 @@ class binary_reader
|
||||
case 0xD2:
|
||||
case 0xD3:
|
||||
case 0xD4:
|
||||
case 0xD5:
|
||||
case 0xD6:
|
||||
case 0xD7:
|
||||
case 0xD8: // tagged item (1 byte follows)
|
||||
case 0xD9: // tagged item (2 bytes follow)
|
||||
case 0xDA: // tagged item (4 bytes follow)
|
||||
@@ -1988,7 +1996,11 @@ class binary_reader
|
||||
{
|
||||
if (get_char)
|
||||
{
|
||||
get(); // TODO(niels): may we ignore N here?
|
||||
// no get_ignore_noop() here: the byte read next must be a string
|
||||
// length type specification, and a no-op ('N') is not valid in
|
||||
// that position. No-ops at positions where a value may appear are
|
||||
// already consumed by the callers via get_ignore_noop().
|
||||
get();
|
||||
}
|
||||
|
||||
if (JSON_HEDLEY_UNLIKELY(!unexpect_eof(input_format, "value")))
|
||||
|
||||
@@ -393,8 +393,12 @@ struct wide_string_input_helper<BaseInputAdapter, 4>
|
||||
}
|
||||
else
|
||||
{
|
||||
// unknown character
|
||||
utf8_bytes[0] = static_cast<std::char_traits<char>::int_type>(wc);
|
||||
// A code point above U+10FFFF has no UTF-8 encoding. Passing the
|
||||
// unit through would narrow it to int, where 0xFFFFFFFF becomes
|
||||
// char_traits<char>::eof() and would end the input silently, so
|
||||
// emit a byte that is never valid UTF-8 and let the decoder
|
||||
// reject it.
|
||||
utf8_bytes[0] = 0xFF;
|
||||
utf8_bytes_filled = 1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -370,8 +370,10 @@ class json_sax_dom_parser
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
// include the length of the quotes, which is 2
|
||||
v.start_position = v.end_position - v.m_data.m_value.string->size() - 2;
|
||||
// escape sequences make the token longer than the value it
|
||||
// parses to, so the start position cannot be derived from
|
||||
// the value; use the offset the lexer recorded instead
|
||||
v.start_position = m_lexer_ref->get_token_start_position();
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -769,8 +771,10 @@ class json_sax_dom_callback_parser
|
||||
|
||||
case value_t::string:
|
||||
{
|
||||
// include the length of the quotes, which is 2
|
||||
v.start_position = v.end_position - v.m_data.m_value.string->size() - 2;
|
||||
// escape sequences make the token longer than the value it
|
||||
// parses to, so the start position cannot be derived from
|
||||
// the value; use the offset the lexer recorded instead
|
||||
v.start_position = m_lexer_ref->get_token_start_position();
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
@@ -1381,6 +1381,11 @@ scan_number_done:
|
||||
token_buffer.clear();
|
||||
decimal_point_position = std::string::npos;
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
// the first character of the token has already been read, hence the -1
|
||||
token_start_position = position.chars_read_total - 1;
|
||||
#endif
|
||||
|
||||
note_token_start(std::integral_constant<bool, lazy_token_string> {});
|
||||
}
|
||||
|
||||
@@ -1581,6 +1586,15 @@ scan_number_done:
|
||||
release_lookahead_impl(std::integral_constant<bool, can_release_lookahead> {});
|
||||
}
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
/// return the offset of the first character of the last read token; unlike
|
||||
/// the token's parsed value, this accounts for escape sequences
|
||||
constexpr std::size_t get_token_start_position() const noexcept
|
||||
{
|
||||
return token_start_position;
|
||||
}
|
||||
#endif
|
||||
|
||||
/// seekable adapter: rebuild the last read token from the input on demand
|
||||
const std::vector<char_type>& collect_token_chars(std::vector<char_type>& out, std::true_type /*lazy*/) const
|
||||
{
|
||||
@@ -1781,6 +1795,12 @@ scan_number_done:
|
||||
/// the last read token on error for seekable adapters (see collect_token_chars)
|
||||
std::size_t token_string_start = 0;
|
||||
|
||||
#if JSON_DIAGNOSTIC_POSITIONS
|
||||
/// start offset of the current token within the input, used to report
|
||||
/// diagnostic positions (see reset())
|
||||
std::size_t token_start_position = 0;
|
||||
#endif
|
||||
|
||||
/// buffer for variable-length tokens (numbers, strings)
|
||||
string_t token_buffer {};
|
||||
|
||||
|
||||
Reference in New Issue
Block a user