mirror of
https://github.com/nlohmann/json.git
synced 2026-10-06 22:47:13 +00:00
`lexer::get()` copied every scanned character into `token_string` on the whole successful-parse hot path, yet that buffer is consumed only by `get_token_string()` when rendering the "last read" fragment of a parse error. On well-formed input the per-byte copy (plus the `unget()` pop) is pure overhead that is always discarded. For seekable input adapters - random-access, single-byte iterators such as those backing `std::string`, `const char*`, and `std::vector<char>` - the offending token is now reconstructed on demand from the input when an error is reported, using a saved start offset, and the eager copy is skipped. Streaming adapters (file, istream, wide-string, and user-defined adapters) keep the eager copy; the strategy is chosen at compile time via `input_adapter_supports_seek`, so adapters without the capability are unaffected. Error messages are byte-for-byte identical across all adapters, verified by a new parity regression test. Microbenchmark (4 MB mixed JSON, parsed from a std::string): ~149 -> ~160 MB/s, about +8%. Signed-off-by: Niels Lohmann <mail@nlohmann.me> Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
1 parent
c034480c22
commit
eed1587000
4 files changed
+369
-16
No files matched your search
@@ -161,8 +161,18 @@ class iterator_input_adapter
|
||||
public:
|
||||
using char_type = typename std::iterator_traits<IteratorType>::value_type;
|
||||
|
||||
// Whether the lexer may reconstruct already-consumed input on demand (for
|
||||
// diagnostics) instead of copying every scanned character eagerly. This is
|
||||
// only sound for multi-pass, randomly-addressable byte input: the iterator
|
||||
// must be random-access (so the consumed prefix can be revisited in O(1))
|
||||
// and each element must map 1:1 to an input byte (wide inputs are wrapped
|
||||
// in wide_string_input_adapter, which does not expose this).
|
||||
static constexpr bool supports_seek =
|
||||
std::is_same<typename std::iterator_traits<IteratorType>::iterator_category, std::random_access_iterator_tag>::value
|
||||
&& sizeof(char_type) == 1;
|
||||
|
||||
iterator_input_adapter(IteratorType first, IteratorType last)
|
||||
: current(std::move(first)), end(std::move(last))
|
||||
: begin(first), current(std::move(first)), end(std::move(last))
|
||||
{}
|
||||
|
||||
typename char_traits<char_type>::int_type get_character()
|
||||
@@ -177,6 +187,22 @@ class iterator_input_adapter
|
||||
return char_traits<char_type>::eof();
|
||||
}
|
||||
|
||||
// number of characters consumed from the input so far
|
||||
std::size_t get_consumed_count() const
|
||||
{
|
||||
return static_cast<std::size_t>(std::distance(begin, current));
|
||||
}
|
||||
|
||||
// append the already-consumed characters in the half-open range
|
||||
// [first_index, last_index) to @a out; only valid when supports_seek
|
||||
template<typename ContainerType>
|
||||
void copy_consumed_range(std::size_t first_index, std::size_t last_index, ContainerType& out) const
|
||||
{
|
||||
const auto from = std::next(begin, static_cast<typename std::iterator_traits<IteratorType>::difference_type>(first_index));
|
||||
const auto to = std::next(begin, static_cast<typename std::iterator_traits<IteratorType>::difference_type>(last_index));
|
||||
out.insert(out.end(), from, to);
|
||||
}
|
||||
|
||||
// for general iterators, we cannot really do something better than falling back to processing the range one-by-one
|
||||
template<class T>
|
||||
std::size_t get_elements(T* dest, std::size_t count = 1)
|
||||
@@ -198,6 +224,7 @@ class iterator_input_adapter
|
||||
}
|
||||
|
||||
private:
|
||||
IteratorType begin;
|
||||
IteratorType current;
|
||||
IteratorType end;
|
||||
|
||||
|
||||
Reference in new issue
Block a user